quantui 0.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. quantui/__init__.py +311 -0
  2. quantui/analytics.py +609 -0
  3. quantui/app.py +5650 -0
  4. quantui/app_analysis.py +662 -0
  5. quantui/app_builders.py +2465 -0
  6. quantui/app_exports.py +194 -0
  7. quantui/app_formatters.py +493 -0
  8. quantui/app_history.py +624 -0
  9. quantui/app_runflow.py +1544 -0
  10. quantui/app_visualization.py +2620 -0
  11. quantui/ase_bridge.py +236 -0
  12. quantui/benchmarks.py +1543 -0
  13. quantui/c_stderr.py +124 -0
  14. quantui/cactus.py +88 -0
  15. quantui/calc_log.py +1116 -0
  16. quantui/calculator.py +204 -0
  17. quantui/cancellation.py +88 -0
  18. quantui/cli.py +288 -0
  19. quantui/comparison.py +306 -0
  20. quantui/config.py +725 -0
  21. quantui/data/js/3Dmol-min.js +2 -0
  22. quantui/data/js/3Dmol-min.js.LICENSE.txt +5 -0
  23. quantui/data/library/library.sqlite +0 -0
  24. quantui/data/manifests/bulk_qm9.json +1 -0
  25. quantui/data/manifests/curated.json +15482 -0
  26. quantui/data/manifests/presets.json +816 -0
  27. quantui/descriptor_cards.py +186 -0
  28. quantui/freq_calc.py +712 -0
  29. quantui/freq_ir_workers.py +229 -0
  30. quantui/gpu_offload.py +278 -0
  31. quantui/help_content.py +474 -0
  32. quantui/ir_plot.py +130 -0
  33. quantui/issue_tracker.py +170 -0
  34. quantui/live_log.py +387 -0
  35. quantui/log_utils.py +492 -0
  36. quantui/molecule.py +577 -0
  37. quantui/molecule_library.py +433 -0
  38. quantui/nmr_calc.py +437 -0
  39. quantui/optimizer.py +670 -0
  40. quantui/orbital_visualization.py +1102 -0
  41. quantui/pes_scan.py +420 -0
  42. quantui/preopt.py +355 -0
  43. quantui/progress.py +111 -0
  44. quantui/pubchem.py +1157 -0
  45. quantui/reorganization_energy.py +435 -0
  46. quantui/results_storage.py +902 -0
  47. quantui/security.py +14 -0
  48. quantui/session_calc.py +622 -0
  49. quantui/structure_providers.py +277 -0
  50. quantui/tddft_calc.py +307 -0
  51. quantui/user_settings.py +238 -0
  52. quantui/utils.py +287 -0
  53. quantui/vib_cache.py +247 -0
  54. quantui/visualization_py3dmol.py +593 -0
  55. quantui/viz_assets.py +101 -0
  56. quantui/viz_backend_router.py +243 -0
  57. quantui-0.5.1.dist-info/METADATA +533 -0
  58. quantui-0.5.1.dist-info/RECORD +62 -0
  59. quantui-0.5.1.dist-info/WHEEL +5 -0
  60. quantui-0.5.1.dist-info/entry_points.txt +2 -0
  61. quantui-0.5.1.dist-info/licenses/LICENSE +21 -0
  62. quantui-0.5.1.dist-info/top_level.txt +1 -0
quantui/benchmarks.py ADDED
@@ -0,0 +1,1543 @@
1
+ """
2
+ Timing calibration benchmark suite for QuantUI.
3
+
4
+ Runs a fixed set of small calculations that span the student-relevant
5
+ method/basis/molecule-size space. Each completed step is logged to
6
+ ``perf_log.jsonl`` via :func:`~quantui.calc_log.log_calculation` so that
7
+ :func:`~quantui.calc_log.estimate_time` immediately becomes useful on a
8
+ fresh install.
9
+
10
+ Four tiers (2026-05-25)
11
+ -----------------------
12
+
13
+ The calibration suite is now a **four-tier cascade** rather than the
14
+ original short/long pair. Users pick the depth that matches their setup-
15
+ time tolerance:
16
+
17
+ - **Tier 1 — Quick** (~15 s): SP only, smoke-test PySCF + bootstrap
18
+ predictor. Same molecules as the historical "short" suite.
19
+ - **Tier 2 — Standard** (~3–5 min): SP only, expanded method × basis
20
+ grid so the predictor has multiple anchors per `(method, basis)` tuple.
21
+ - **Tier 3 — Mixed** (~10–15 min): tier 2 + 2–3 small geometry
22
+ optimizations + 1–2 small frequency calcs. First reliable GeoOpt +
23
+ Freq predictions.
24
+ - **Tier 4 — Deep** (up to 30 min): tier 3 + medium GeoOpt + medium
25
+ Freq (ethanol, benzene) + MP2 / CCSD anchors. Lets the estimator
26
+ predict every calc-type × device combo within ±25%.
27
+
28
+ Back-compat: the legacy ``mode="short"`` / ``mode="long"`` strings still
29
+ work as aliases for tier 1 / tier 2 respectively. New code should use
30
+ ``mode="tier1"`` … ``mode="tier4"``.
31
+
32
+ Entry format
33
+ ------------
34
+
35
+ Each tier is a list of 7-tuples (single-point calcs) or 8-tuples (when
36
+ the 8th element overrides the calc-type, e.g. ``"geometry_opt"`` /
37
+ ``"frequency"``). ``_normalize_entry()`` unpacks either shape.
38
+
39
+ Typical usage (from the UI)::
40
+
41
+ import threading
42
+ from quantui.benchmarks import run_calibration
43
+
44
+ stop = threading.Event()
45
+ result = run_calibration(
46
+ progress_cb=lambda *a: print(a),
47
+ stop_event=stop,
48
+ timeout_per_step=120,
49
+ mode="tier3", # or "tier1"/"tier2"/"tier4"
50
+ )
51
+ """
52
+
53
+ from __future__ import annotations
54
+
55
+ import time
56
+ from dataclasses import dataclass, field
57
+ from datetime import datetime, timezone
58
+ from pathlib import Path
59
+ from typing import Callable, List, Optional
60
+
61
+ # ---------------------------------------------------------------------------
62
+ # Benchmark suite definition
63
+ # ---------------------------------------------------------------------------
64
+
65
+ #: Each entry: (label, atoms, coordinates, charge, multiplicity, method, basis)
66
+ #: Molecules are kept deliberately small so the full suite finishes quickly on
67
+ #: any modern laptop.
68
+ BENCHMARK_SUITE: list[tuple] = [
69
+ (
70
+ "H₂ RHF/STO-3G",
71
+ ["H", "H"],
72
+ [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]],
73
+ 0,
74
+ 1,
75
+ "RHF",
76
+ "STO-3G",
77
+ ),
78
+ (
79
+ "H₂O RHF/STO-3G",
80
+ ["O", "H", "H"],
81
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
82
+ 0,
83
+ 1,
84
+ "RHF",
85
+ "STO-3G",
86
+ ),
87
+ (
88
+ "H₂O B3LYP/STO-3G",
89
+ ["O", "H", "H"],
90
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
91
+ 0,
92
+ 1,
93
+ "B3LYP",
94
+ "STO-3G",
95
+ ),
96
+ (
97
+ "H₂O RHF/6-31G*",
98
+ ["O", "H", "H"],
99
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
100
+ 0,
101
+ 1,
102
+ "RHF",
103
+ "6-31G*",
104
+ ),
105
+ (
106
+ "CH₄ RHF/STO-3G",
107
+ ["C", "H", "H", "H", "H"],
108
+ [
109
+ [0.0, 0.0, 0.0],
110
+ [0.629, 0.629, 0.629],
111
+ [-0.629, -0.629, 0.629],
112
+ [-0.629, 0.629, -0.629],
113
+ [0.629, -0.629, -0.629],
114
+ ],
115
+ 0,
116
+ 1,
117
+ "RHF",
118
+ "STO-3G",
119
+ ),
120
+ (
121
+ "C₂H₄ RHF/STO-3G",
122
+ ["C", "C", "H", "H", "H", "H"],
123
+ [
124
+ [0.0, 0.0, 0.670],
125
+ [0.0, 0.0, -0.670],
126
+ [0.0, 0.924, 1.241],
127
+ [0.0, -0.924, 1.241],
128
+ [0.0, 0.924, -1.241],
129
+ [0.0, -0.924, -1.241],
130
+ ],
131
+ 0,
132
+ 1,
133
+ "RHF",
134
+ "STO-3G",
135
+ ),
136
+ (
137
+ "C₂H₆O (ethanol) RHF/STO-3G",
138
+ ["C", "C", "O", "H", "H", "H", "H", "H", "H"],
139
+ [
140
+ [-1.232, 0.026, 0.000],
141
+ [0.281, 0.026, 0.000],
142
+ [0.829, 1.310, 0.000],
143
+ [-1.566, 1.059, 0.000],
144
+ [-1.609, -0.506, 0.880],
145
+ [-1.609, -0.506, -0.880],
146
+ [0.668, -0.497, 0.890],
147
+ [0.668, -0.497, -0.890],
148
+ [1.802, 1.311, 0.000],
149
+ ],
150
+ 0,
151
+ 1,
152
+ "RHF",
153
+ "STO-3G",
154
+ ),
155
+ (
156
+ "C₂H₆O (ethanol) B3LYP/6-31G*",
157
+ ["C", "C", "O", "H", "H", "H", "H", "H", "H"],
158
+ [
159
+ [-1.232, 0.026, 0.000],
160
+ [0.281, 0.026, 0.000],
161
+ [0.829, 1.310, 0.000],
162
+ [-1.566, 1.059, 0.000],
163
+ [-1.609, -0.506, 0.880],
164
+ [-1.609, -0.506, -0.880],
165
+ [0.668, -0.497, 0.890],
166
+ [0.668, -0.497, -0.890],
167
+ [1.802, 1.311, 0.000],
168
+ ],
169
+ 0,
170
+ 1,
171
+ "B3LYP",
172
+ "6-31G*",
173
+ ),
174
+ ]
175
+
176
+ #: Extended suite for a full calibration run (~3–6 min on a modern laptop).
177
+ #: Includes the short suite plus larger molecules and more expensive methods
178
+ #: to anchor the efficiency model across the student-relevant size range.
179
+ BENCHMARK_SUITE_LONG: list[tuple] = [
180
+ *BENCHMARK_SUITE,
181
+ # ── Additional entries ─────────────────────────────────────────────────
182
+ (
183
+ "H₂O RHF/cc-pVDZ",
184
+ ["O", "H", "H"],
185
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
186
+ 0,
187
+ 1,
188
+ "RHF",
189
+ "cc-pVDZ",
190
+ ),
191
+ (
192
+ "C₂H₆O (ethanol) RHF/6-31G*",
193
+ ["C", "C", "O", "H", "H", "H", "H", "H", "H"],
194
+ [
195
+ [-1.232, 0.026, 0.000],
196
+ [0.281, 0.026, 0.000],
197
+ [0.829, 1.310, 0.000],
198
+ [-1.566, 1.059, 0.000],
199
+ [-1.609, -0.506, 0.880],
200
+ [-1.609, -0.506, -0.880],
201
+ [0.668, -0.497, 0.890],
202
+ [0.668, -0.497, -0.890],
203
+ [1.802, 1.311, 0.000],
204
+ ],
205
+ 0,
206
+ 1,
207
+ "RHF",
208
+ "6-31G*",
209
+ ),
210
+ (
211
+ "C₆H₆ (benzene) RHF/STO-3G",
212
+ ["C", "C", "C", "C", "C", "C", "H", "H", "H", "H", "H", "H"],
213
+ [
214
+ [1.395, 0.000, 0.000],
215
+ [0.698, 1.209, 0.000],
216
+ [-0.698, 1.209, 0.000],
217
+ [-1.395, 0.000, 0.000],
218
+ [-0.698, -1.209, 0.000],
219
+ [0.698, -1.209, 0.000],
220
+ [2.479, 0.000, 0.000],
221
+ [1.240, 2.147, 0.000],
222
+ [-1.240, 2.147, 0.000],
223
+ [-2.479, 0.000, 0.000],
224
+ [-1.240, -2.147, 0.000],
225
+ [1.240, -2.147, 0.000],
226
+ ],
227
+ 0,
228
+ 1,
229
+ "RHF",
230
+ "STO-3G",
231
+ ),
232
+ (
233
+ "C₆H₆ (benzene) RHF/6-31G*",
234
+ ["C", "C", "C", "C", "C", "C", "H", "H", "H", "H", "H", "H"],
235
+ [
236
+ [1.395, 0.000, 0.000],
237
+ [0.698, 1.209, 0.000],
238
+ [-0.698, 1.209, 0.000],
239
+ [-1.395, 0.000, 0.000],
240
+ [-0.698, -1.209, 0.000],
241
+ [0.698, -1.209, 0.000],
242
+ [2.479, 0.000, 0.000],
243
+ [1.240, 2.147, 0.000],
244
+ [-1.240, 2.147, 0.000],
245
+ [-2.479, 0.000, 0.000],
246
+ [-1.240, -2.147, 0.000],
247
+ [1.240, -2.147, 0.000],
248
+ ],
249
+ 0,
250
+ 1,
251
+ "RHF",
252
+ "6-31G*",
253
+ ),
254
+ (
255
+ "C₆H₆ (benzene) B3LYP/6-31G*",
256
+ ["C", "C", "C", "C", "C", "C", "H", "H", "H", "H", "H", "H"],
257
+ [
258
+ [1.395, 0.000, 0.000],
259
+ [0.698, 1.209, 0.000],
260
+ [-0.698, 1.209, 0.000],
261
+ [-1.395, 0.000, 0.000],
262
+ [-0.698, -1.209, 0.000],
263
+ [0.698, -1.209, 0.000],
264
+ [2.479, 0.000, 0.000],
265
+ [1.240, 2.147, 0.000],
266
+ [-1.240, 2.147, 0.000],
267
+ [-2.479, 0.000, 0.000],
268
+ [-1.240, -2.147, 0.000],
269
+ [1.240, -2.147, 0.000],
270
+ ],
271
+ 0,
272
+ 1,
273
+ "B3LYP",
274
+ "6-31G*",
275
+ ),
276
+ (
277
+ "C₁₀H₈ (naphthalene) RHF/STO-3G",
278
+ [
279
+ "C",
280
+ "C",
281
+ "C",
282
+ "C",
283
+ "C",
284
+ "C",
285
+ "C",
286
+ "C",
287
+ "C",
288
+ "C",
289
+ "H",
290
+ "H",
291
+ "H",
292
+ "H",
293
+ "H",
294
+ "H",
295
+ "H",
296
+ "H",
297
+ ],
298
+ [
299
+ [1.243, 1.400, 0.000],
300
+ [2.440, 0.725, 0.000],
301
+ [2.440, -0.725, 0.000],
302
+ [1.243, -1.400, 0.000],
303
+ [0.000, -0.720, 0.000],
304
+ [0.000, 0.720, 0.000],
305
+ [-1.243, 1.400, 0.000],
306
+ [-2.440, 0.725, 0.000],
307
+ [-2.440, -0.725, 0.000],
308
+ [-1.243, -1.400, 0.000],
309
+ [1.237, 2.488, 0.000],
310
+ [3.377, 1.244, 0.000],
311
+ [3.377, -1.244, 0.000],
312
+ [1.237, -2.488, 0.000],
313
+ [-1.237, -2.488, 0.000],
314
+ [-3.377, -1.244, 0.000],
315
+ [-3.377, 1.244, 0.000],
316
+ [-1.237, 2.488, 0.000],
317
+ ],
318
+ 0,
319
+ 1,
320
+ "RHF",
321
+ "STO-3G",
322
+ ),
323
+ # ── Expansion (2026-05-25) ────────────────────────────────────────────
324
+ # Additional SP entries that broaden the method × basis grid coverage,
325
+ # extending tier 2's expected wall-clock to the 3-5 min target.
326
+ (
327
+ "H₂O B3LYP/6-31G*",
328
+ ["O", "H", "H"],
329
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
330
+ 0,
331
+ 1,
332
+ "B3LYP",
333
+ "6-31G*",
334
+ ),
335
+ (
336
+ "H₂O wB97X-D/6-31G*",
337
+ ["O", "H", "H"],
338
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
339
+ 0,
340
+ 1,
341
+ "wB97X-D",
342
+ "6-31G*",
343
+ ),
344
+ (
345
+ "CH₄ B3LYP/6-31G*",
346
+ ["C", "H", "H", "H", "H"],
347
+ [
348
+ [0.0, 0.0, 0.0],
349
+ [0.629, 0.629, 0.629],
350
+ [-0.629, -0.629, 0.629],
351
+ [-0.629, 0.629, -0.629],
352
+ [0.629, -0.629, -0.629],
353
+ ],
354
+ 0,
355
+ 1,
356
+ "B3LYP",
357
+ "6-31G*",
358
+ ),
359
+ (
360
+ "NH₃ RHF/cc-pVDZ",
361
+ ["N", "H", "H", "H"],
362
+ [
363
+ [0.000, 0.000, 0.111],
364
+ [0.000, 0.940, -0.260],
365
+ [0.814, -0.470, -0.260],
366
+ [-0.814, -0.470, -0.260],
367
+ ],
368
+ 0,
369
+ 1,
370
+ "RHF",
371
+ "cc-pVDZ",
372
+ ),
373
+ (
374
+ "NH₃ B3LYP/cc-pVDZ",
375
+ ["N", "H", "H", "H"],
376
+ [
377
+ [0.000, 0.000, 0.111],
378
+ [0.000, 0.940, -0.260],
379
+ [0.814, -0.470, -0.260],
380
+ [-0.814, -0.470, -0.260],
381
+ ],
382
+ 0,
383
+ 1,
384
+ "B3LYP",
385
+ "cc-pVDZ",
386
+ ),
387
+ (
388
+ "H₂CO (formaldehyde) B3LYP/6-31G*",
389
+ ["C", "O", "H", "H"],
390
+ [
391
+ [0.000, 0.000, 0.000],
392
+ [0.000, 0.000, 1.207],
393
+ [0.000, 0.943, -0.589],
394
+ [0.000, -0.943, -0.589],
395
+ ],
396
+ 0,
397
+ 1,
398
+ "B3LYP",
399
+ "6-31G*",
400
+ ),
401
+ ]
402
+
403
+
404
+ # ---------------------------------------------------------------------------
405
+ # Tier 3 — Mixed (~10-15 min): tier 2 + small GeoOpts + small Freqs
406
+ # ---------------------------------------------------------------------------
407
+ #
408
+ # 8-tuple entries override the default ``"single_point"`` calc-type. The 8th
409
+ # element is one of ``"geometry_opt"`` / ``"frequency"``.
410
+ #
411
+ # Small geometry opts (3-5 atoms) and the cheapest realistic frequency calc
412
+ # (H₂O / B3LYP / STO-3G) anchor the multi-calc-type predictions without
413
+ # blowing the time budget.
414
+
415
+ BENCHMARK_SUITE_TIER3: list[tuple] = [
416
+ *BENCHMARK_SUITE_LONG,
417
+ # ── Small GeoOpts ─────────────────────────────────────────────────────
418
+ (
419
+ "H₂O B3LYP/STO-3G [GeoOpt]",
420
+ ["O", "H", "H"],
421
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
422
+ 0,
423
+ 1,
424
+ "B3LYP",
425
+ "STO-3G",
426
+ "geometry_opt",
427
+ ),
428
+ (
429
+ "H₂CO B3LYP/6-31G* [GeoOpt]",
430
+ ["C", "O", "H", "H"],
431
+ [
432
+ [0.000, 0.000, 0.000],
433
+ [0.000, 0.000, 1.207],
434
+ [0.000, 0.943, -0.589],
435
+ [0.000, -0.943, -0.589],
436
+ ],
437
+ 0,
438
+ 1,
439
+ "B3LYP",
440
+ "6-31G*",
441
+ "geometry_opt",
442
+ ),
443
+ (
444
+ "CH₄ B3LYP/6-31G* [GeoOpt]",
445
+ ["C", "H", "H", "H", "H"],
446
+ [
447
+ [0.0, 0.0, 0.0],
448
+ [0.629, 0.629, 0.629],
449
+ [-0.629, -0.629, 0.629],
450
+ [-0.629, 0.629, -0.629],
451
+ [0.629, -0.629, -0.629],
452
+ ],
453
+ 0,
454
+ 1,
455
+ "B3LYP",
456
+ "6-31G*",
457
+ "geometry_opt",
458
+ ),
459
+ # ── Small Freqs (cheapest realistic anchors for the 6N inner-SCF model) ──
460
+ (
461
+ "H₂O B3LYP/STO-3G [Freq]",
462
+ ["O", "H", "H"],
463
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
464
+ 0,
465
+ 1,
466
+ "B3LYP",
467
+ "STO-3G",
468
+ "frequency",
469
+ ),
470
+ (
471
+ "H₂CO B3LYP/6-31G* [Freq]",
472
+ ["C", "O", "H", "H"],
473
+ [
474
+ [0.000, 0.000, 0.000],
475
+ [0.000, 0.000, 1.207],
476
+ [0.000, 0.943, -0.589],
477
+ [0.000, -0.943, -0.589],
478
+ ],
479
+ 0,
480
+ 1,
481
+ "B3LYP",
482
+ "6-31G*",
483
+ "frequency",
484
+ ),
485
+ ]
486
+
487
+
488
+ # ---------------------------------------------------------------------------
489
+ # Tier 4 — Deep (up to 30 min): tier 3 + medium GeoOpt + medium Freq + MP2/CCSD
490
+ # ---------------------------------------------------------------------------
491
+ #
492
+ # Medium-size geometry opt + medium-size frequency anchors the predictor
493
+ # across realistic molecule sizes. MP2 + CCSD entries on H₂O / cc-pVDZ
494
+ # anchor the β=5.0 (MP2) and β=6.0 (CCSD) scaling exponents in
495
+ # ``calc_log._METHOD_SCALE_EXP``. The benzene frequency is the workhorse
496
+ # parallel-IR test — 12 atoms × 6 = 72 inner SCFs.
497
+
498
+ BENCHMARK_SUITE_TIER4: list[tuple] = [
499
+ *BENCHMARK_SUITE_TIER3,
500
+ # ── Medium GeoOpt ─────────────────────────────────────────────────────
501
+ (
502
+ "C₂H₆O (ethanol) B3LYP/6-31G* [GeoOpt]",
503
+ ["C", "C", "O", "H", "H", "H", "H", "H", "H"],
504
+ [
505
+ [-1.232, 0.026, 0.000],
506
+ [0.281, 0.026, 0.000],
507
+ [0.829, 1.310, 0.000],
508
+ [-1.566, 1.059, 0.000],
509
+ [-1.609, -0.506, 0.880],
510
+ [-1.609, -0.506, -0.880],
511
+ [0.668, -0.497, 0.890],
512
+ [0.668, -0.497, -0.890],
513
+ [1.802, 1.311, 0.000],
514
+ ],
515
+ 0,
516
+ 1,
517
+ "B3LYP",
518
+ "6-31G*",
519
+ "geometry_opt",
520
+ ),
521
+ # ── Medium Freq ───────────────────────────────────────────────────────
522
+ (
523
+ "C₂H₆O (ethanol) B3LYP/6-31G* [Freq]",
524
+ ["C", "C", "O", "H", "H", "H", "H", "H", "H"],
525
+ [
526
+ [-1.232, 0.026, 0.000],
527
+ [0.281, 0.026, 0.000],
528
+ [0.829, 1.310, 0.000],
529
+ [-1.566, 1.059, 0.000],
530
+ [-1.609, -0.506, 0.880],
531
+ [-1.609, -0.506, -0.880],
532
+ [0.668, -0.497, 0.890],
533
+ [0.668, -0.497, -0.890],
534
+ [1.802, 1.311, 0.000],
535
+ ],
536
+ 0,
537
+ 1,
538
+ "B3LYP",
539
+ "6-31G*",
540
+ "frequency",
541
+ ),
542
+ (
543
+ "C₆H₆ (benzene) B3LYP/6-31G* [Freq]",
544
+ ["C", "C", "C", "C", "C", "C", "H", "H", "H", "H", "H", "H"],
545
+ [
546
+ [1.395, 0.000, 0.000],
547
+ [0.698, 1.209, 0.000],
548
+ [-0.698, 1.209, 0.000],
549
+ [-1.395, 0.000, 0.000],
550
+ [-0.698, -1.209, 0.000],
551
+ [0.698, -1.209, 0.000],
552
+ [2.479, 0.000, 0.000],
553
+ [1.240, 2.147, 0.000],
554
+ [-1.240, 2.147, 0.000],
555
+ [-2.479, 0.000, 0.000],
556
+ [-1.240, -2.147, 0.000],
557
+ [1.240, -2.147, 0.000],
558
+ ],
559
+ 0,
560
+ 1,
561
+ "B3LYP",
562
+ "6-31G*",
563
+ "frequency",
564
+ ),
565
+ # ── Post-HF anchors ───────────────────────────────────────────────────
566
+ (
567
+ "H₂O MP2/cc-pVDZ",
568
+ ["O", "H", "H"],
569
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
570
+ 0,
571
+ 1,
572
+ "MP2",
573
+ "cc-pVDZ",
574
+ ),
575
+ (
576
+ "H₂O CCSD/cc-pVDZ",
577
+ ["O", "H", "H"],
578
+ [[0.0, 0.0, 0.0], [0.757, 0.587, 0.0], [-0.757, 0.587, 0.0]],
579
+ 0,
580
+ 1,
581
+ "CCSD",
582
+ "cc-pVDZ",
583
+ ),
584
+ ]
585
+
586
+
587
+ # Aliases — keep BENCHMARK_SUITE / BENCHMARK_SUITE_LONG for back-compat
588
+ # (existing tests + app.py imports). New code should reference the
589
+ # tier-named constants for clarity.
590
+ BENCHMARK_SUITE_TIER1: list[tuple] = BENCHMARK_SUITE
591
+ BENCHMARK_SUITE_TIER2: list[tuple] = BENCHMARK_SUITE_LONG
592
+
593
+
594
+ # ---------------------------------------------------------------------------
595
+ # Mode-string → suite mapping
596
+ # ---------------------------------------------------------------------------
597
+ #
598
+ # ``run_calibration(mode=)`` accepts any of these strings. The legacy
599
+ # ``"short"`` / ``"long"`` aliases are kept so older callers (including
600
+ # pinned UI state) keep working.
601
+
602
+ _MODE_TO_SUITE: dict = {
603
+ "tier1": BENCHMARK_SUITE_TIER1,
604
+ "tier2": BENCHMARK_SUITE_TIER2,
605
+ "tier3": BENCHMARK_SUITE_TIER3,
606
+ "tier4": BENCHMARK_SUITE_TIER4,
607
+ "short": BENCHMARK_SUITE_TIER1,
608
+ "long": BENCHMARK_SUITE_TIER2,
609
+ }
610
+
611
+
612
+ def _normalize_entry(entry: tuple) -> dict:
613
+ """Unpack a 7-tuple or 8-tuple benchmark entry into a uniform dict.
614
+
615
+ 7-tuple: ``(label, atoms, coords, charge, mult, method, basis)`` —
616
+ defaults ``calc_type`` to ``"single_point"``.
617
+
618
+ 8-tuple: ``(label, atoms, coords, charge, mult, method, basis, calc_type)``
619
+ — used by tier 3 + tier 4 entries that need ``"geometry_opt"`` or
620
+ ``"frequency"`` dispatch.
621
+ """
622
+ if len(entry) == 7:
623
+ label, atoms, coords, charge, mult, method, basis = entry
624
+ calc_type = "single_point"
625
+ elif len(entry) == 8:
626
+ label, atoms, coords, charge, mult, method, basis, calc_type = entry
627
+ else:
628
+ raise ValueError(
629
+ f"Benchmark entry must have 7 or 8 fields, got {len(entry)}: {entry!r}"
630
+ )
631
+ return {
632
+ "label": label,
633
+ "atoms": atoms,
634
+ "coords": coords,
635
+ "charge": charge,
636
+ "multiplicity": mult,
637
+ "method": method,
638
+ "basis": basis,
639
+ "calc_type": calc_type,
640
+ }
641
+
642
+
643
+ # ---------------------------------------------------------------------------
644
+ # Cross-device probe (2026-05-25)
645
+ # ---------------------------------------------------------------------------
646
+ #
647
+ # When GPU offload is available, tier 3 and tier 4 calibrations should run
648
+ # a SMALL representative subset of entries twice — once on GPU and once on
649
+ # CPU (via ``QUANTUI_DISABLE_GPU=1``) — so a single calibration populates
650
+ # the analytics dashboard's GPU-vs-CPU speedup table with measured pairs
651
+ # rather than asking users to re-run the suite under different env vars.
652
+ #
653
+ # Doubling the WHOLE tier would blow the time budget (tier 4 is already
654
+ # up to 30 min); 3-4 representative entries per tier costs ~5-10 min
655
+ # extra on a GPU host and is the right granularity for the speedup table.
656
+
657
+ #: Labels of benchmark entries that get a CPU/GPU probe pair in tier 3+4.
658
+ #: Matched exactly against the ``label`` field of normalized entries. Keep
659
+ #: this short — one cheap SP, one medium SP, one cheap freq is plenty.
660
+ _CROSS_DEVICE_PROBE_LABELS = frozenset(
661
+ {
662
+ "H₂O B3LYP/6-31G*",
663
+ "C₆H₆ (benzene) B3LYP/6-31G*",
664
+ "H₂O B3LYP/STO-3G [Freq]",
665
+ }
666
+ )
667
+
668
+
669
+ def _build_execution_plan(suite: list, mode: str, gpu_available: bool) -> list[dict]:
670
+ """Expand the suite into a list of execution entries.
671
+
672
+ Each entry is a normalized dict with an additional ``force_cpu``
673
+ bool field. Non-probe entries appear once with ``force_cpu=False``.
674
+ Probe entries appear:
675
+
676
+ - **once** when GPU is unavailable or the tier is 1/2 (no cross-
677
+ device data to collect).
678
+ - **twice** when GPU is available AND mode is tier3/tier4 — once
679
+ with ``force_cpu=False`` (will use GPU offload) and once with
680
+ ``force_cpu=True`` (will set ``QUANTUI_DISABLE_GPU=1`` in the
681
+ worker's environment). Labels are suffixed ``[GPU]`` / ``[CPU]``
682
+ to keep the results table unambiguous.
683
+
684
+ The worker reads ``force_cpu`` and toggles the env var BEFORE any
685
+ quantui / gpu4pyscf import so the cached probe sees the right state.
686
+ """
687
+ do_cross_device = gpu_available and mode in ("tier3", "tier4")
688
+ plan: list[dict] = []
689
+ for entry in suite:
690
+ normalized = _normalize_entry(entry)
691
+ if do_cross_device and normalized["label"] in _CROSS_DEVICE_PROBE_LABELS:
692
+ gpu_variant = dict(normalized)
693
+ gpu_variant["label"] = f"{normalized['label']} [GPU]"
694
+ gpu_variant["force_cpu"] = False
695
+ cpu_variant = dict(normalized)
696
+ cpu_variant["label"] = f"{normalized['label']} [CPU]"
697
+ cpu_variant["force_cpu"] = True
698
+ plan.append(gpu_variant)
699
+ plan.append(cpu_variant)
700
+ else:
701
+ normalized["force_cpu"] = False
702
+ plan.append(normalized)
703
+ return plan
704
+
705
+
706
+ # ---------------------------------------------------------------------------
707
+ # Result dataclass
708
+ # ---------------------------------------------------------------------------
709
+
710
+ _STATUS_OK = "ok"
711
+ _STATUS_TIMEOUT = "timed_out"
712
+ _STATUS_STOPPED = "stopped" # whole-suite stop (e.g. Stop button)
713
+ _STATUS_SKIPPED = "skipped" # single-step skip (e.g. Skip button)
714
+ _STATUS_ERROR = "error"
715
+
716
+
717
+ @dataclass
718
+ class BenchmarkStep:
719
+ """Result for a single benchmark step."""
720
+
721
+ label: str
722
+ method: str
723
+ basis: str
724
+ n_atoms: int
725
+ n_electrons: int
726
+ status: str # "ok" | "timed_out" | "stopped" | "error"
727
+ elapsed_s: float = 0.0
728
+ error_msg: str = ""
729
+ n_basis: Optional[int] = None
730
+ # Track which calc-type this step ran so tier 3+4
731
+ # entries can be distinguished in summaries.
732
+ calc_type: str = "single_point"
733
+ # Follow-up (2026-05-25): the calibration worker
734
+ # now saves each step as a real result directory (via save_result)
735
+ # so users can re-open them from the History tab like any other
736
+ # calc. ``None`` when save_result failed (best-effort) or the step
737
+ # itself errored before completion.
738
+ result_dir: Optional[str] = None
739
+
740
+
741
+ @dataclass
742
+ class CalibrationResult:
743
+ """Summary result from :func:`run_calibration`."""
744
+
745
+ timestamp: str
746
+ steps: List[BenchmarkStep] = field(default_factory=list)
747
+ stopped_early: bool = False
748
+ mode: str = "tier1"
749
+ # The cross-device probe expands the execution plan beyond
750
+ # ``len(_MODE_TO_SUITE[mode])`` for tier 3/4 on GPU hosts. Store
751
+ # the plan length explicitly so progress denominators stay correct;
752
+ # 0 (default) means "fall back to suite size" for back-compat with
753
+ # callers that construct the dataclass directly without a runner.
754
+ expected_steps: int = 0
755
+
756
+ @property
757
+ def n_completed(self) -> int:
758
+ return sum(1 for s in self.steps if s.status == _STATUS_OK)
759
+
760
+ @property
761
+ def n_total(self) -> int:
762
+ if self.expected_steps:
763
+ return self.expected_steps
764
+ return len(_MODE_TO_SUITE.get(self.mode, BENCHMARK_SUITE_TIER1))
765
+
766
+
767
+ # ---------------------------------------------------------------------------
768
+ # Main calibration runner
769
+ # ---------------------------------------------------------------------------
770
+
771
+ ProgressCallback = Callable[[int, int, str, str, float], None]
772
+ """progress_cb(step_n, total, label, status, elapsed_s)"""
773
+
774
+
775
+ def _count_electrons(atoms: list[str], charge: int) -> int:
776
+ """Rough electron count: sum of atomic numbers minus charge."""
777
+ from .config import ATOMIC_NUMBERS
778
+
779
+ return sum(ATOMIC_NUMBERS.get(a, 6) for a in atoms) - charge
780
+
781
+
782
+ # ---------------------------------------------------------------------------
783
+ # Subprocess worker (2026-05-25)
784
+ # ---------------------------------------------------------------------------
785
+ #
786
+ # Originally calibration ran each step in a ThreadPoolExecutor with a
787
+ # ``future.result(timeout=...)`` block. That had three blockers exposed by
788
+ # a tier-4 attempt:
789
+ #
790
+ # 1. The Stop button only checked between steps, so an in-flight 5-minute
791
+ # freq calc could not be killed mid-run.
792
+ # 2. There was no per-step progress signal beyond a single "running"
793
+ # label — the user couldn't tell whether a slow step had frozen the
794
+ # kernel.
795
+ # 3. ``calibration.json`` was only flushed at the END of the loop, so
796
+ # stopping at step 25/30 lost the partial-state marker.
797
+ #
798
+ # The fix runs each step in a child process via ``multiprocessing.Process``
799
+ # so ``worker.terminate()`` works reliably cross-platform. The worker pipes
800
+ # PySCF's progress stream to a calibration log file the main process tails
801
+ # every 500 ms for the live status display, and ``calibration.json`` is
802
+ # rewritten after each completed step.
803
+
804
+
805
+ class _TeeStream:
806
+ """Minimal text stream that fans writes to multiple destinations.
807
+
808
+ Used in the calibration worker so PySCF's ``progress_stream`` output
809
+ lands BOTH in the shared per-run calibration log (for the parent's
810
+ live tail) AND in an in-memory ``StringIO`` (so we can pass the
811
+ per-calc PySCF log text to ``save_result`` for the result dir's
812
+ ``pyscf.log`` file). Errors writing to any one stream are swallowed
813
+ — the goal is never to take down the calc because of a bad fanout.
814
+ """
815
+
816
+ def __init__(self, *streams) -> None:
817
+ self._streams = streams
818
+
819
+ def write(self, s) -> int:
820
+ for stream in self._streams:
821
+ try:
822
+ stream.write(s)
823
+ except Exception: # noqa: BLE001 — tee best-effort
824
+ pass
825
+ return len(s)
826
+
827
+ def flush(self) -> None:
828
+ for stream in self._streams:
829
+ try:
830
+ stream.flush()
831
+ except Exception: # noqa: BLE001 — tee best-effort
832
+ pass
833
+
834
+
835
+ def _save_calibration_step(
836
+ res,
837
+ *,
838
+ calc_type: str,
839
+ pyscf_log: str,
840
+ calibration_run_id: str,
841
+ mol,
842
+ ):
843
+ """Save a completed calibration calc as a regular result directory.
844
+
845
+ Matches the save sequence from ``_do_run`` in ``app.py`` so the
846
+ History browser can load + replay calibration entries like any
847
+ user-initiated calc:
848
+
849
+ - ``save_result`` — base dir + result.json + pyscf.log. The
850
+ ``extras={"calibration_run_id": ...}`` tag lets the History
851
+ dropdown render a 🔧 marker beside calibration entries.
852
+ - ``save_thumbnail`` — the card shown in the History dropdown.
853
+ - For GeoOpt: ``save_trajectory`` (so the Trajectory panel works).
854
+ - For SP/GeoOpt/Freq with MO data: ``save_orbitals`` (so the
855
+ Energies + Isosurface panels work).
856
+ - For Freq: a ``spectra`` dict baked into result.json so the IR
857
+ + Vibrational panels work; ``displacements`` serialized to
858
+ nested lists.
859
+
860
+ Returns the result directory path, or ``None`` on save failure
861
+ (caller treats this as "calc succeeded but couldn't save — log it
862
+ but don't fail the step").
863
+ """
864
+ from quantui.results_storage import (
865
+ load_result,
866
+ save_orbitals,
867
+ save_result,
868
+ save_thumbnail,
869
+ save_trajectory,
870
+ )
871
+
872
+ # Build the spectra dict for Frequency calcs — must match what the
873
+ # Analysis tab's _pop_ir_spectrum / _pop_vibrational expect.
874
+ spectra: dict = {}
875
+ if calc_type == "frequency":
876
+ displacements_serialized = None
877
+ try:
878
+ import numpy as _np
879
+
880
+ if getattr(res, "displacements", None) is not None:
881
+ displacements_serialized = _np.asarray(res.displacements).tolist()
882
+ except Exception: # noqa: BLE001 — best-effort
883
+ pass
884
+ spectra = {
885
+ "ir": {
886
+ "frequencies_cm1": getattr(res, "frequencies_cm1", []),
887
+ "ir_intensities": getattr(res, "ir_intensities", []),
888
+ "zpve_hartree": getattr(res, "zpve_hartree", 0.0),
889
+ "displacements": displacements_serialized,
890
+ },
891
+ "molecule": {
892
+ "atoms": list(mol.atoms),
893
+ "coords": [list(map(float, row)) for row in mol.coordinates],
894
+ "charge": mol.charge,
895
+ "multiplicity": mol.multiplicity,
896
+ },
897
+ }
898
+
899
+ # For GeoOpt the ``res`` from optimize_geometry has its own .method /
900
+ # .basis / .formula via res.molecule. save_result expects those
901
+ # attributes on the top-level result. Build a uniform shim.
902
+ if calc_type == "geometry_opt":
903
+ from types import SimpleNamespace
904
+
905
+ save_obj = SimpleNamespace(
906
+ formula=res.molecule.get_formula(),
907
+ method=res.method,
908
+ basis=res.basis,
909
+ energy_hartree=(
910
+ res.energies_hartree[-1] if res.energies_hartree else float("nan")
911
+ ),
912
+ converged=bool(res.converged),
913
+ n_iterations=int(getattr(res, "n_steps", -1)),
914
+ homo_lumo_gap_ev=None,
915
+ mo_energy_hartree=getattr(res, "mo_energy_hartree", None),
916
+ mo_occ=getattr(res, "mo_occ", None),
917
+ mo_coeff=getattr(res, "mo_coeff", None),
918
+ pyscf_mol_atom=getattr(res, "pyscf_mol_atom", None),
919
+ pyscf_mol_basis=getattr(res, "pyscf_mol_basis", None),
920
+ )
921
+ else:
922
+ save_obj = res
923
+
924
+ extras = {"calibration_run_id": calibration_run_id}
925
+ try:
926
+ saved_dir = save_result(
927
+ save_obj,
928
+ pyscf_log=pyscf_log,
929
+ calc_type=calc_type,
930
+ spectra=spectra or None,
931
+ extras=extras,
932
+ )
933
+ except Exception: # noqa: BLE001 — save is best-effort
934
+ return None
935
+
936
+ # Best-effort follow-on saves. None of these are required for the
937
+ # History card to render — they enrich the replay experience.
938
+ try:
939
+ saved_data = load_result(saved_dir)
940
+ save_thumbnail(saved_dir, saved_data)
941
+ except Exception: # noqa: BLE001 — thumbnail is purely cosmetic
942
+ pass
943
+
944
+ if calc_type == "geometry_opt":
945
+ try:
946
+ traj = getattr(res, "trajectory", None) or getattr(res, "molecule", None)
947
+ energies = list(getattr(res, "energies_hartree", []) or [])
948
+ if traj and not isinstance(traj, list):
949
+ traj = [traj]
950
+ if traj and len(traj) >= 1:
951
+ save_trajectory(saved_dir, traj, energies)
952
+ except Exception: # noqa: BLE001 — trajectory save is best-effort
953
+ pass
954
+
955
+ if calc_type in ("single_point", "geometry_opt", "frequency"):
956
+ try:
957
+ save_orbitals(saved_dir, save_obj)
958
+ except Exception: # noqa: BLE001 — orbital save is best-effort
959
+ pass
960
+
961
+ return saved_dir
962
+
963
+
964
+ def _calibration_worker(
965
+ atoms: list,
966
+ coords: list,
967
+ charge: int,
968
+ mult: int,
969
+ method: str,
970
+ basis: str,
971
+ calc_type: str,
972
+ log_path_str: str,
973
+ result_queue,
974
+ calibration_run_id: str = "",
975
+ force_cpu: bool = False,
976
+ ) -> None:
977
+ """Run one calibration step in a child process.
978
+
979
+ Picklable (top-level function, primitive args + a Queue). Pipes
980
+ PySCF progress to ``log_path_str`` (append mode) so the parent can
981
+ tail it AND to an in-memory buffer so the per-calc PySCF output
982
+ can be saved alongside the result.
983
+
984
+ ``force_cpu=True`` sets ``QUANTUI_DISABLE_GPU=1`` in the worker's
985
+ environment BEFORE any quantui / gpu4pyscf import so the cached
986
+ ``is_gpu_available()`` probe sees the override and the calc actually
987
+ runs on CPU. Used by the cross-device probe — tier 3/4 on a
988
+ GPU host runs selected entries twice (once forced-CPU, once GPU) so
989
+ the analytics speedup table is populated from one calibration run.
990
+
991
+ On success: saves a real result directory via ``_save_calibration_step``
992
+ (tagged with ``calibration_run_id``) and puts a summary dict with
993
+ ``result_dir`` on ``result_queue``.
994
+
995
+ On exception: puts ``{"status": "error", "error_msg": ..., "result_dir": None}``.
996
+ The parent treats absence of a queue entry (after worker exit) as a
997
+ crashed worker — distinct from a step-level error.
998
+ """
999
+ import io as _io
1000
+ import os as _os
1001
+ import time as _t
1002
+ from datetime import datetime as _dt
1003
+ from pathlib import Path as _P
1004
+
1005
+ # Must run BEFORE any quantui / pyscf / gpu4pyscf import so
1006
+ # the ``is_gpu_available()`` cache sees the override on first probe.
1007
+ if force_cpu:
1008
+ _os.environ["QUANTUI_DISABLE_GPU"] = "1"
1009
+
1010
+ log_path = _P(log_path_str)
1011
+ t0 = _t.perf_counter()
1012
+ label = f"{method}/{basis} ({calc_type})"
1013
+
1014
+ try:
1015
+ # Line-buffered append so the parent's tail sees output as it
1016
+ # arrives. ``buffering=1`` requires text mode (which we use).
1017
+ # The tee fans writes to both the shared log + an in-memory
1018
+ # buffer so we can save the per-calc PySCF output to the
1019
+ # result dir's pyscf.log.
1020
+ with open(log_path, "a", encoding="utf-8", buffering=1) as log_fh:
1021
+ log_fh.write(
1022
+ f"\n========= {_dt.utcnow().isoformat()} :: {label} =========\n"
1023
+ )
1024
+ per_calc_buf = _io.StringIO()
1025
+ stream = _TeeStream(log_fh, per_calc_buf)
1026
+
1027
+ from quantui.molecule import Molecule as _Molecule
1028
+
1029
+ mol = _Molecule(atoms, coords, charge=charge, multiplicity=mult)
1030
+
1031
+ if calc_type == "geometry_opt":
1032
+ from quantui.optimizer import optimize_geometry as _opt
1033
+
1034
+ res = _opt(
1035
+ molecule=mol,
1036
+ method=method,
1037
+ basis=basis,
1038
+ progress_stream=stream,
1039
+ )
1040
+ formula = res.molecule.get_formula()
1041
+ converged = bool(res.converged)
1042
+ n_iterations = int(getattr(res, "n_steps", -1))
1043
+ elif calc_type == "frequency":
1044
+ from quantui.freq_calc import run_freq_calc as _freq
1045
+
1046
+ res = _freq(
1047
+ molecule=mol,
1048
+ method=method,
1049
+ basis=basis,
1050
+ progress_stream=stream,
1051
+ )
1052
+ formula = res.formula
1053
+ converged = bool(res.converged)
1054
+ n_iterations = int(res.n_iterations)
1055
+ else: # single_point
1056
+ from quantui.session_calc import run_in_session as _sp
1057
+
1058
+ # verbose=3 gives per-iteration SCF energies in the log —
1059
+ # enough signal to confirm the worker hasn't frozen on a
1060
+ # slow tier-4 entry. (Was verbose=0 previously.)
1061
+ res = _sp(
1062
+ mol,
1063
+ method=method,
1064
+ basis=basis,
1065
+ verbose=3,
1066
+ progress_stream=stream,
1067
+ )
1068
+ formula = res.formula
1069
+ converged = bool(res.converged)
1070
+ n_iterations = int(res.n_iterations)
1071
+
1072
+ elapsed = _t.perf_counter() - t0
1073
+ log_fh.write(f"\n[QuantUI_STATUS] COMPLETED in {elapsed:.2f} s\n")
1074
+
1075
+ # Save as a regular result directory (2026-05-25 — tier 4's
1076
+ # MP2 + CCSD + benzene freq are scientifically valuable;
1077
+ # don't discard them).
1078
+ saved_dir = _save_calibration_step(
1079
+ res,
1080
+ calc_type=calc_type,
1081
+ pyscf_log=per_calc_buf.getvalue(),
1082
+ calibration_run_id=calibration_run_id,
1083
+ mol=mol,
1084
+ )
1085
+
1086
+ result_queue.put(
1087
+ {
1088
+ "status": "ok",
1089
+ "formula": formula,
1090
+ "converged": converged,
1091
+ "n_iterations": n_iterations,
1092
+ "elapsed_s": elapsed,
1093
+ "result_dir": str(saved_dir) if saved_dir else None,
1094
+ }
1095
+ )
1096
+ except Exception as exc:
1097
+ result_queue.put(
1098
+ {
1099
+ "status": "error",
1100
+ "error_msg": str(exc)[:500],
1101
+ "elapsed_s": _t.perf_counter() - t0,
1102
+ "result_dir": None,
1103
+ }
1104
+ )
1105
+
1106
+
1107
+ def _tail_last_status_line(log_path) -> str:
1108
+ """Return the last meaningful progress line from the calibration log.
1109
+
1110
+ Prefers ``[QuantUI_STATUS] ...`` markers emitted by ``freq_calc``;
1111
+ falls back to any non-blank line. Truncated to ~120 chars so the
1112
+ UI widget renders cleanly. Returns "" on any IO failure (best-
1113
+ effort).
1114
+ """
1115
+ try:
1116
+ with open(log_path, encoding="utf-8", errors="replace") as fh:
1117
+ lines = fh.readlines()
1118
+ except OSError:
1119
+ return ""
1120
+ # Walk backwards looking for the best candidate.
1121
+ status_line = ""
1122
+ fallback_line = ""
1123
+ for line in reversed(lines):
1124
+ stripped = line.strip()
1125
+ if not stripped:
1126
+ continue
1127
+ if "[QuantUI_STATUS]" in stripped:
1128
+ status_line = stripped
1129
+ break
1130
+ if not fallback_line:
1131
+ fallback_line = stripped
1132
+ best = status_line or fallback_line
1133
+ if len(best) > 120:
1134
+ best = best[-120:]
1135
+ return best
1136
+
1137
+
1138
+ def _calibration_log_path(timestamp: str) -> Path:
1139
+ """Return the path to the per-run calibration log file.
1140
+
1141
+ Filename includes the run timestamp so multiple runs don't clobber
1142
+ each other. Lives under ``~/.quantui/logs/`` (honours
1143
+ ``QUANTUI_LOG_DIR``) alongside the event + perf logs.
1144
+ """
1145
+ import os as _os
1146
+
1147
+ env = _os.environ.get("QUANTUI_LOG_DIR")
1148
+ base = Path(env) if env else Path.home() / ".quantui" / "logs"
1149
+ # Make a filename-safe timestamp.
1150
+ safe_ts = timestamp.replace(":", "-").replace(".", "-")
1151
+ return base / f"calibration_{safe_ts}.log"
1152
+
1153
+
1154
+ def _save_calibration_json(result: CalibrationResult, log_path: Path) -> None:
1155
+ """Persist the current ``CalibrationResult`` snapshot to disk.
1156
+
1157
+ Called after EVERY completed step (not just at end-of-run) so an
1158
+ interrupted tier-4 still records the partial-state marker the user
1159
+ can see next session. Includes the log file path so the "last
1160
+ calibration" UI can link to the per-run log.
1161
+ """
1162
+ import json as _json
1163
+
1164
+ cal_path = Path.home() / ".quantui" / "calibration.json"
1165
+ try:
1166
+ cal_path.parent.mkdir(parents=True, exist_ok=True)
1167
+ cal_path.write_text(
1168
+ _json.dumps(
1169
+ {
1170
+ "timestamp": result.timestamp,
1171
+ "mode": result.mode,
1172
+ "stopped_early": result.stopped_early,
1173
+ "log_path": str(log_path),
1174
+ "n_completed": result.n_completed,
1175
+ "n_total": result.n_total,
1176
+ "steps": [
1177
+ {
1178
+ "label": s.label,
1179
+ "method": s.method,
1180
+ "basis": s.basis,
1181
+ "n_atoms": s.n_atoms,
1182
+ "n_electrons": s.n_electrons,
1183
+ "n_basis": s.n_basis,
1184
+ "status": s.status,
1185
+ "elapsed_s": round(s.elapsed_s, 3),
1186
+ "error_msg": s.error_msg,
1187
+ "calc_type": s.calc_type,
1188
+ "result_dir": s.result_dir,
1189
+ }
1190
+ for s in result.steps
1191
+ ],
1192
+ },
1193
+ indent=2,
1194
+ ensure_ascii=False,
1195
+ ),
1196
+ encoding="utf-8",
1197
+ )
1198
+ except OSError:
1199
+ # Disk full / permission denied — best-effort. The perf log is
1200
+ # the canonical record; calibration.json is just a UI summary.
1201
+ pass
1202
+
1203
+
1204
+ def run_calibration(
1205
+ progress_cb: Optional[ProgressCallback] = None,
1206
+ stop_event=None,
1207
+ timeout_per_step: Optional[float] = None,
1208
+ mode: str = "tier1",
1209
+ skip_event=None,
1210
+ ) -> CalibrationResult:
1211
+ """Run the benchmark suite and populate ``perf_log.jsonl``.
1212
+
1213
+ Each step runs in a child process so the Stop button can terminate
1214
+ a long-running calc mid-run. Per-step progress is piped to a log
1215
+ file under ``~/.quantui/logs/calibration_<timestamp>.log`` and the
1216
+ parent tails it every 500 ms to drive the live status display.
1217
+ ``~/.quantui/calibration.json`` is rewritten after every completed
1218
+ step, so an interrupted run still records partial state.
1219
+
1220
+ Args:
1221
+ progress_cb: Called periodically with
1222
+ ``(step_n, total, label, status, elapsed_s)`` and optionally
1223
+ ``live_message=<latest log line>`` during slow steps. The
1224
+ terminal call after each step uses status in
1225
+ ``ok / timed_out / stopped / skipped / error``; intermediate
1226
+ "running" ticks fire while the step is in-flight.
1227
+ stop_event: A :class:`threading.Event`; checked every 500 ms.
1228
+ When set, the in-flight worker is terminated immediately,
1229
+ the current step is marked ``"stopped"``, and remaining
1230
+ steps are abandoned (no further work).
1231
+ timeout_per_step: Wall-clock seconds allowed per step.
1232
+ ``None`` (default) means no timeout — the user controls
1233
+ stoppage via the Stop / Skip buttons. A tier-4
1234
+ run had a benzene B3LYP/6-31G* freq calc finish at
1235
+ ~1500 s but be cut off at the old 1800 s hard cap, losing
1236
+ the data; the no-timeout default removes that hazard.
1237
+ Pass a numeric value only when running headlessly (e.g. CI)
1238
+ where you genuinely want a wall-clock cap.
1239
+ mode: One of ``"tier1"`` / ``"tier2"`` / ``"tier3"`` / ``"tier4"``.
1240
+ Legacy aliases ``"short"`` / ``"long"`` map to tier1 / tier2.
1241
+ Unknown modes fall back to tier1 with a warning.
1242
+ skip_event: A :class:`threading.Event`; checked every 500 ms.
1243
+ When set, the in-flight worker is terminated, the current
1244
+ step is marked ``"skipped"``, the event is cleared, and
1245
+ the loop continues to the NEXT step. Distinct from
1246
+ ``stop_event``: skip is one step, stop is the whole run.
1247
+
1248
+ Returns:
1249
+ :class:`CalibrationResult` with per-step outcomes.
1250
+ """
1251
+ import multiprocessing as _mp
1252
+ import queue as _queue
1253
+
1254
+ from quantui import calc_log as _calc_log
1255
+
1256
+ _pyscf_available = False
1257
+ try:
1258
+ import pyscf # noqa: F401
1259
+
1260
+ _pyscf_available = True
1261
+ except ImportError:
1262
+ pass
1263
+
1264
+ if mode not in _MODE_TO_SUITE:
1265
+ import logging as _log
1266
+
1267
+ _log.getLogger(__name__).warning(
1268
+ "run_calibration: unknown mode %r, falling back to tier1", mode
1269
+ )
1270
+ mode = "tier1"
1271
+ suite = _MODE_TO_SUITE[mode]
1272
+
1273
+ # Probe GPU availability once in the parent so we know whether
1274
+ # to duplicate cross-device entries. Failure (e.g. gpu_offload import
1275
+ # error on a misconfigured install) defaults to "no GPU" — the
1276
+ # calibration still runs, it just doesn't collect speedup pairs.
1277
+ gpu_available = False
1278
+ try:
1279
+ from quantui.gpu_offload import is_gpu_available as _is_gpu_avail
1280
+
1281
+ gpu_available = bool(_is_gpu_avail()[0])
1282
+ except Exception: # noqa: BLE001 — best-effort probe
1283
+ gpu_available = False
1284
+
1285
+ execution_plan = _build_execution_plan(suite, mode, gpu_available)
1286
+ timestamp = datetime.now(timezone.utc).isoformat()
1287
+ total = len(execution_plan)
1288
+ result = CalibrationResult(timestamp=timestamp, mode=mode, expected_steps=total)
1289
+
1290
+ # Per-run calibration log file. The worker appends; the parent tails.
1291
+ log_path = _calibration_log_path(timestamp)
1292
+ timeout_str = (
1293
+ f"{timeout_per_step:.0f} s"
1294
+ if timeout_per_step is not None
1295
+ else "none (user-controlled)"
1296
+ )
1297
+ try:
1298
+ log_path.parent.mkdir(parents=True, exist_ok=True)
1299
+ with open(log_path, "w", encoding="utf-8") as fh:
1300
+ fh.write(
1301
+ f"QuantUI calibration log\n"
1302
+ f"started : {timestamp}\n"
1303
+ f"mode : {mode}\n"
1304
+ f"suite size: {total} entries\n"
1305
+ f"timeout/step: {timeout_str}\n"
1306
+ )
1307
+ except OSError:
1308
+ # No log file is non-fatal — calibration still runs, just without
1309
+ # the per-step progress trail.
1310
+ pass
1311
+
1312
+ # Use ``spawn`` everywhere: ``fork`` from a
1313
+ # background thread (run_calibration runs inside ``_do_calibration``
1314
+ # which is itself a daemon thread) collides hard with CUDA contexts
1315
+ # that the parent process may have initialized via the GPU-detection
1316
+ # probe — every step would die at ~0.04 s with no useful error.
1317
+ # ``spawn`` adds ~1-2 s startup overhead per step but isolates the
1318
+ # worker from the parent's interpreter state entirely, so CUDA / MPI /
1319
+ # any C-extension global is freshly initialized. Sub-2-second-per-step
1320
+ # overhead is a great trade for "the Stop button works AND nothing
1321
+ # crashes for opaque reasons".
1322
+ _ctx = _mp.get_context("spawn")
1323
+
1324
+ def _emit_progress(*args, live_message=None, step=None) -> None:
1325
+ """Wrap progress_cb to tolerate callers that pre-date the
1326
+ ``live_message`` / ``step`` kwargs (notably the test-suite
1327
+ lambdas that accept ``*args`` only). Falls back through each
1328
+ new kwarg in turn on ``TypeError``."""
1329
+ if progress_cb is None:
1330
+ return
1331
+ # Try newest signature first, peel off kwargs the caller can't
1332
+ # accept. Modern callers (do_calibration) take both; tests pass
1333
+ # ``lambda *a: ...``.
1334
+ try:
1335
+ progress_cb(*args, live_message=live_message, step=step)
1336
+ return
1337
+ except TypeError:
1338
+ pass
1339
+ try:
1340
+ progress_cb(*args, live_message=live_message)
1341
+ return
1342
+ except TypeError:
1343
+ pass
1344
+ progress_cb(*args)
1345
+
1346
+ stopped_mid_step = False
1347
+ for step_n, normalized in enumerate(execution_plan, start=1):
1348
+ label = normalized["label"]
1349
+ atoms = normalized["atoms"]
1350
+ coords = normalized["coords"]
1351
+ charge = normalized["charge"]
1352
+ mult = normalized["multiplicity"]
1353
+ method = normalized["method"]
1354
+ basis = normalized["basis"]
1355
+ calc_type = normalized["calc_type"]
1356
+ force_cpu = bool(normalized.get("force_cpu", False))
1357
+
1358
+ # Honour stop request BEFORE starting a new step.
1359
+ if stop_event is not None and stop_event.is_set():
1360
+ result.stopped_early = True
1361
+ break
1362
+
1363
+ nb = _calc_log.count_basis_functions(atoms, basis)
1364
+ step = BenchmarkStep(
1365
+ label=label,
1366
+ method=method,
1367
+ basis=basis,
1368
+ n_atoms=len(atoms),
1369
+ n_electrons=_count_electrons(atoms, charge),
1370
+ status=_STATUS_ERROR,
1371
+ n_basis=nb,
1372
+ calc_type=calc_type,
1373
+ )
1374
+
1375
+ if not _pyscf_available:
1376
+ step.error_msg = "PySCF not available"
1377
+ result.steps.append(step)
1378
+ _save_calibration_json(result, log_path)
1379
+ _emit_progress(step_n, total, label, step.status, 0.0, step=step)
1380
+ continue
1381
+
1382
+ # Spawn the worker.
1383
+ result_queue = _ctx.Queue()
1384
+ worker = _ctx.Process(
1385
+ target=_calibration_worker,
1386
+ args=(
1387
+ atoms,
1388
+ coords,
1389
+ charge,
1390
+ mult,
1391
+ method,
1392
+ basis,
1393
+ calc_type,
1394
+ str(log_path),
1395
+ result_queue,
1396
+ timestamp, # calibration_run_id — the parent's run timestamp
1397
+ force_cpu, # cross-device probe flag
1398
+ ),
1399
+ daemon=True,
1400
+ )
1401
+ t_start = time.perf_counter()
1402
+ worker.start()
1403
+
1404
+ # Poll loop — finish naturally OR hit timeout OR stop OR skip.
1405
+ poll_interval = 0.5
1406
+ worker_done_normally = False
1407
+ while True:
1408
+ worker.join(timeout=poll_interval)
1409
+ elapsed = time.perf_counter() - t_start
1410
+
1411
+ if not worker.is_alive():
1412
+ worker_done_normally = True
1413
+ break
1414
+
1415
+ # Timeout is now opt-in (was a hard 1800 s for tier 4 which
1416
+ # cut off a near-finishing benzene freq).
1417
+ # ``None`` means "user controls; never auto-kill".
1418
+ if timeout_per_step is not None and elapsed > timeout_per_step:
1419
+ worker.terminate()
1420
+ worker.join(timeout=5)
1421
+ step.status = _STATUS_TIMEOUT
1422
+ step.elapsed_s = elapsed
1423
+ step.error_msg = f"exceeded {timeout_per_step:.0f}s timeout"
1424
+ break
1425
+
1426
+ if stop_event is not None and stop_event.is_set():
1427
+ worker.terminate()
1428
+ worker.join(timeout=5)
1429
+ step.status = _STATUS_STOPPED
1430
+ step.elapsed_s = elapsed
1431
+ result.stopped_early = True
1432
+ stopped_mid_step = True
1433
+ break
1434
+
1435
+ # Skip = "abandon THIS step, continue to the next." Distinct
1436
+ # from Stop. Clear the event after consuming so the next
1437
+ # step starts fresh — the UI re-sets it if the user clicks
1438
+ # Skip again. (Replaces the hard timeout that was cutting
1439
+ # off near-finishing calcs.)
1440
+ if skip_event is not None and skip_event.is_set():
1441
+ worker.terminate()
1442
+ worker.join(timeout=5)
1443
+ step.status = _STATUS_SKIPPED
1444
+ step.elapsed_s = elapsed
1445
+ step.error_msg = f"skipped by user at {elapsed:.0f}s"
1446
+ skip_event.clear()
1447
+ break
1448
+
1449
+ # Live-tick: pull the latest log line for the UI.
1450
+ live_msg = _tail_last_status_line(log_path)
1451
+ _emit_progress(
1452
+ step_n, total, label, "running", elapsed, live_message=live_msg
1453
+ )
1454
+
1455
+ if worker_done_normally:
1456
+ try:
1457
+ msg = result_queue.get(timeout=2.0)
1458
+ except _queue.Empty:
1459
+ # Worker process exited (either crashed during import,
1460
+ # raised before reaching the worker's try/except, or
1461
+ # was killed by the OS) without putting anything on
1462
+ # the queue. Capture the exit code + the tail of the
1463
+ # calibration log so the user can see what actually
1464
+ # happened — "worker exited without result" alone is
1465
+ # useless for diagnosis (the original symptom of every
1466
+ # step failing at 0.04 s).
1467
+ _exitcode = getattr(worker, "exitcode", None)
1468
+ _tail = _tail_last_status_line(log_path) or "(no log output)"
1469
+ _hint = ""
1470
+ if _exitcode is not None and _exitcode != 0:
1471
+ # On Unix, negative exit codes encode the signal
1472
+ # that killed the process (-9 = SIGKILL, -11 = SEGV).
1473
+ if _exitcode < 0:
1474
+ import signal as _sig
1475
+
1476
+ try:
1477
+ _sig_name = _sig.Signals(-_exitcode).name
1478
+ _hint = f" (killed by {_sig_name})"
1479
+ except (ValueError, AttributeError):
1480
+ _hint = f" (signal {-_exitcode})"
1481
+ msg = {
1482
+ "status": "error",
1483
+ "error_msg": (
1484
+ f"worker exited (exitcode={_exitcode}){_hint}; "
1485
+ f"last log line: {_tail}"
1486
+ )[:500],
1487
+ "elapsed_s": time.perf_counter() - t_start,
1488
+ }
1489
+ if msg.get("status") == "ok":
1490
+ step.status = _STATUS_OK
1491
+ step.elapsed_s = float(msg["elapsed_s"])
1492
+ step.result_dir = msg.get("result_dir")
1493
+ # Log to perf_log.jsonl so estimate_time() picks it up.
1494
+ _calc_log.log_calculation(
1495
+ formula=msg["formula"],
1496
+ n_atoms=step.n_atoms,
1497
+ n_electrons=step.n_electrons,
1498
+ method=method,
1499
+ basis=basis,
1500
+ n_iterations=int(msg.get("n_iterations", -1)),
1501
+ elapsed_s=float(msg["elapsed_s"]),
1502
+ converged=bool(msg["converged"]),
1503
+ n_basis=step.n_basis,
1504
+ n_cores=1,
1505
+ calc_type=calc_type,
1506
+ )
1507
+ else:
1508
+ step.status = _STATUS_ERROR
1509
+ step.error_msg = msg.get("error_msg", "unknown")
1510
+ step.elapsed_s = float(
1511
+ msg.get("elapsed_s", time.perf_counter() - t_start)
1512
+ )
1513
+
1514
+ result.steps.append(step)
1515
+ # Persist after EVERY step so an interrupt at step N
1516
+ # still leaves a partial-state record on disk.
1517
+ _save_calibration_json(result, log_path)
1518
+
1519
+ # Terminal call for this step — pass the full BenchmarkStep so
1520
+ # the UI callback can append it to the incremental results table.
1521
+ _emit_progress(step_n, total, label, step.status, step.elapsed_s, step=step)
1522
+
1523
+ if stopped_mid_step:
1524
+ break
1525
+
1526
+ # Final write (idempotent — same content as the last per-step write
1527
+ # unless the loop broke via the top-of-loop stop check).
1528
+ _save_calibration_json(result, log_path)
1529
+ return result
1530
+
1531
+
1532
+ def load_last_calibration() -> Optional[dict]:
1533
+ """Return the last calibration summary dict, or ``None`` if absent."""
1534
+ import json
1535
+
1536
+ path = Path.home() / ".quantui" / "calibration.json"
1537
+ if not path.exists():
1538
+ return None
1539
+ try:
1540
+ data: dict = json.loads(path.read_text(encoding="utf-8"))
1541
+ return data
1542
+ except Exception:
1543
+ return None