From 0a1508b6608a83f7ba07294de40251869d82f279 Mon Sep 17 00:00:00 2001 From: Schultz Lab at NCCU Date: Sun, 13 Sep 2026 00:23:42 -0400 Subject: [PATCH 1/2] feat: add guarded GPU support for geometry engines --- README.md | 27 +++++- docs/CLI.md | 6 +- pyproject.toml | 12 +++ quantui/app.py | 13 ++- quantui/backends/worker_payload.py | 2 + quantui/cli.py | 55 ++++++++---- quantui/engines/base.py | 7 +- quantui/engines/pyfock_engine.py | 31 ++++++- quantui/engines/pyscf_engine.py | 5 ++ quantui/optimizer.py | 76 ++++++++++++++-- quantui/pyfock_gpu.py | 118 +++++++++++++++++++++++++ tests/test_cli.py | 14 +++ tests/test_gpu_geometry_integration.py | 76 ++++++++++++++++ tests/test_optimizer.py | 23 +++++ tests/test_pyfock_engine.py | 11 +++ tests/test_pyfock_gpu.py | 60 +++++++++++++ 16 files changed, 504 insertions(+), 32 deletions(-) create mode 100644 quantui/pyfock_gpu.py create mode 100644 tests/test_gpu_geometry_integration.py create mode 100644 tests/test_pyfock_gpu.py diff --git a/README.md b/README.md index 6782cb3..0edfaad 100644 --- a/README.md +++ b/README.md @@ -145,8 +145,10 @@ selected when PySCF is absent). The validated subset supports neutral, closed-shell PBE single points and geometry optimizations with def2-SVP or def2-TZVP. Density fitting is always enabled, analytical gradients drive optimization, and orbital, Mulliken, dipole, and cube analysis are retained. -Hybrids, charged/open-shell systems, solvent, checkpoint warm starts, and GPU -remain PySCF-only. +Hybrids, charged/open-shell systems, solvent, and checkpoint warm starts remain +PySCF-only. Optional PyFock GPU acceleration is available separately through +CuPy; geometry optimization uses numerical forces on GPU because PyFock 0.1.7 +does not yet provide analytical GPU gradients. PySCF does not install on Windows natively. For the complete feature set, the [`apptainer/quantui.def`](https://github.com/The-Schultz-Lab/QuantUI/blob/main/apptainer/quantui.def) container bundles @@ -240,6 +242,27 @@ and result cards will display the compute device. Whenever gpu4pyscf can't offload a particular call, QuantUI falls back to CPU automatically and the result card reflects which device ran. +### Optional: PyFock GPU acceleration + +PyFock uses its own CuPy/Numba CUDA implementation; it does not use +`gpu4pyscf`. Install the PyFock engine and the CUDA-suffixed CuPy extra that +matches the NVIDIA driver reported by `nvidia-smi`: + +```bash +# CUDA 13.x driver +pip install "quantui[pyfock,pyfock-gpu-cuda13x]" + +# CUDA 12.x driver +pip install "quantui[pyfock,pyfock-gpu-cuda12x]" +``` + +PyFock GPU use is guarded independently: QuantUI requires CuPy to import and +report a CUDA device, and still honors the Settings GPU toggle and +`QUANTUI_DISABLE_GPU=1`. If the probe fails, PyFock falls back to CPU and +reports the reason. GPU single points use PyFock's GPU SCF/integral/XC path; +GPU geometry optimizations use numerical finite-difference forces because the +installed PyFock release's analytical gradient implementation is CPU-only. + ### Optional: GFN-FF metal pre-optimization (xtb) The classical (MMFF/UFF) pre-optimizer relies on RDKit's organic valence diff --git a/docs/CLI.md b/docs/CLI.md index ad78caa..2d2ff82 100644 --- a/docs/CLI.md +++ b/docs/CLI.md @@ -177,7 +177,8 @@ quantui log tail -n 200 | grep -i error | tail -5 Probe whether QuantUI's GPU offload path is functional in the current environment. This is the canonical one-liner for verifying that `gpu4pyscf` + `cupy` are installed correctly and that -`is_gpu_available()` will return `True` when the app runs. +`is_gpu_available()` will return `True` when the PySCF app path runs. Use +`--engine pyfock` to probe PyFock's independent CuPy path. ### Flags @@ -189,6 +190,9 @@ None. # Is GPU offload working right now? quantui gpu check +# Check the separate PyFock/CuPy path +quantui gpu check --engine pyfock + # Use in a shell condition if quantui gpu check; then echo "GPU mode" diff --git a/pyproject.toml b/pyproject.toml index bb5d169..4d5eea7 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -190,6 +190,16 @@ gpu-cuda13x = [ "cutensor-cu13", ] +# PyFock GPU acceleration is a separate CuPy-only path. Do not install or +# probe gpu4pyscf for a PyFock calculation: the two engines have independent +# CUDA implementations and capability gates. +pyfock-gpu-cuda12x = [ + "cupy-cuda12x", +] +pyfock-gpu-cuda13x = [ + "cupy-cuda13x", +] + # Notebook smoke-test dependencies notebook = [ "nbmake>=1.4.0", @@ -329,6 +339,8 @@ markers = [ "network: marks tests that require network connectivity", "notebook: marks notebook smoke tests", "pyfock: marks real PyFock integration/parity tests (opt in with QUANTUI_RUN_PYFOCK_INTEGRATION=1)", + "gpu_integration: marks real NVIDIA GPU integration tests (opt in with QUANTUI_RUN_GPU_INTEGRATION=1)", + "pyfock_gpu: marks real PyFock CuPy GPU tests (opt in with QUANTUI_RUN_PYFOCK_GPU_INTEGRATION=1)", ] [tool.coverage.run] diff --git a/quantui/app.py b/quantui/app.py index f513eeb..13df928 100644 --- a/quantui/app.py +++ b/quantui/app.py @@ -5812,7 +5812,11 @@ def _run_required_final_single_point(target_mol, reason: str): "atoms": list(target_mol.atoms), "coordinates": [list(c) for c in target_mol.coordinates], }, - options={"verbose": 4, "scf_rescue": True}, + options={ + "verbose": 4, + "scf_rescue": True, + "use_gpu": bool(self._user_settings.compute.gpu_enabled), + }, progress_stream=log, # type: ignore[arg-type] solvent=_solvent, ), @@ -6007,6 +6011,7 @@ def _run_required_final_single_point(target_mol, reason: str): ), "resume": _resume, "scf_rescue": True, + "use_gpu": bool(self._user_settings.compute.gpu_enabled), }, progress_stream=log, # type: ignore[arg-type] checkpoint=_ckpt, @@ -6430,7 +6435,11 @@ def _run_required_final_single_point(target_mol, reason: str): "atoms": list(calc_mol.atoms), "coordinates": [list(c) for c in calc_mol.coordinates], }, - options={"verbose": 4, "scf_rescue": True}, + options={ + "verbose": 4, + "scf_rescue": True, + "use_gpu": bool(self._user_settings.compute.gpu_enabled), + }, progress_stream=log, # type: ignore[arg-type] solvent=_solvent, checkpoint=_ckpt, diff --git a/quantui/backends/worker_payload.py b/quantui/backends/worker_payload.py index 80e4b32..3ded223 100644 --- a/quantui/backends/worker_payload.py +++ b/quantui/backends/worker_payload.py @@ -116,6 +116,8 @@ def optimization_result_payload(result, *, trajectory_file: str) -> Dict[str, An "method": result.method, "basis": result.basis, "formula": result.formula, + "gpu_used": bool(getattr(result, "gpu_used", False)), + "gpu_name": getattr(result, "gpu_name", None), "trajectory_file": trajectory_file, } diff --git a/quantui/cli.py b/quantui/cli.py index 6b27ba3..a9dffad 100644 --- a/quantui/cli.py +++ b/quantui/cli.py @@ -9,9 +9,9 @@ * ``quantui log tail [-n N]`` — print the last N event-log entries (default 20). Reads ``~/.quantui/logs/event_log.jsonl`` honoring the ``QUANTUI_LOG_DIR`` env override. -* ``quantui gpu check`` — run QuantUI's GPU-offload detection and print - ``(available, device-name)``. Exit code 0 when GPU is usable, 1 when - not — handy for one-line CI / shell-script gating. +* ``quantui gpu check [--engine pyscf|pyfock]`` — run one engine's GPU + detection and print ``(available, device-name)``. Exit code 0 when GPU is + usable, 1 when not — handy for one-line CI / shell-script gating. * ``quantui analytics build [-o PATH] [--open]`` — build a self-contained HTML analytics dashboard from ``perf_log.jsonl``. Default output: ``~/.quantui/dashboard.html``. Pass ``--open`` to automatically open @@ -106,19 +106,36 @@ def _cmd_gpu_check(args: argparse.Namespace) -> int: Returns exit code 0 when GPU offload is available, 1 when it's not — so ``if quantui gpu check; then ...; fi`` works in shell scripts. """ - from quantui.gpu_offload import is_gpu_available, is_low_fp64_device, probe_gpu - - # The detection probe is cached; clear so each CLI invocation is - # fresh (the user may have just installed gpu4pyscf and wants to - # confirm without restarting their shell). - # cache_clear is forwarded from _probe_gpu's lru_cache onto this function - # at definition time (gpu_offload.py); mypy can't see a monkey-patched - # attribute across the module boundary. - is_gpu_available.cache_clear() # type: ignore[attr-defined] - available, name, reason = probe_gpu() + engine = getattr(args, "engine", "pyscf") + if engine == "pyfock": + from quantui.pyfock_gpu import ( + clear_pyfock_gpu_probe_cache, + probe_pyfock_gpu, + ) + + clear_pyfock_gpu_probe_cache() + available, name, reason = probe_pyfock_gpu() + label = "PyFock GPU" + low_fp64 = False + else: + from quantui.gpu_offload import ( + is_gpu_available, + is_low_fp64_device, + probe_gpu, + ) + + # The detection probe is cached; clear so each CLI invocation is + # fresh (the user may have just installed gpu4pyscf and wants to + # confirm without restarting their shell). + # cache_clear is forwarded from _probe_gpu's lru_cache onto this + # function at definition time; mypy can't see that attribute. + is_gpu_available.cache_clear() # type: ignore[attr-defined] + available, name, reason = probe_gpu() + label = "GPU offload" + low_fp64 = is_low_fp64_device(name) if available: - print(f"GPU offload available: {name}") - if is_low_fp64_device(name): + print(f"{label} available: {name}") + if low_fp64: # Available is not the same as worth using: PySCF is FP64 # throughout, and consumer cards gate double precision to a small # fraction of single. Say so here rather than let the user discover @@ -391,7 +408,13 @@ def _build_parser() -> argparse.ArgumentParser: gpu_sub = gpu_parser.add_subparsers(dest="gpu_command", required=True) gpu_check = gpu_sub.add_parser( "check", - help="Run QuantUI's GPU-offload detection probe.", + help="Run a GPU detection probe.", + ) + gpu_check.add_argument( + "--engine", + choices=("pyscf", "pyfock"), + default="pyscf", + help="Engine-specific GPU path to probe (default: pyscf).", ) gpu_check.set_defaults(func=_cmd_gpu_check) diff --git a/quantui/engines/base.py b/quantui/engines/base.py index 8ddd8bb..b53d2b6 100644 --- a/quantui/engines/base.py +++ b/quantui/engines/base.py @@ -1,7 +1,8 @@ """ Quantum-engine contract types (v0.1). -See ``QuantUI-development-tracking/TODO/QUANTUM-ENGINE-CONTRACT.md``. +The engine contract is maintained alongside the project's development +documentation. """ from __future__ import annotations @@ -85,6 +86,8 @@ class EngineResult: homo_lumo_gap_ev: Optional[float] = None warnings: List[str] = field(default_factory=list) error: Optional[Dict[str, Any]] = None + gpu_used: bool = False + gpu_name: Optional[str] = None native_result: Optional[Any] = field(default=None, repr=False, compare=False) def to_dict(self) -> Dict[str, Any]: @@ -118,6 +121,8 @@ def to_session_result(self) -> Any: basis=self.basis, formula=self.formula, density_fit=self.engine_id == "pyfock", + gpu_used=self.gpu_used, + gpu_name=self.gpu_name, scf_variant="RKS" if self.engine_id == "pyfock" else "", engine_id=self.engine_id, ) diff --git a/quantui/engines/pyfock_engine.py b/quantui/engines/pyfock_engine.py index c683dcc..69fb16d 100644 --- a/quantui/engines/pyfock_engine.py +++ b/quantui/engines/pyfock_engine.py @@ -7,6 +7,7 @@ from contextlib import redirect_stderr, redirect_stdout from typing import Any, Optional +from ..pyfock_gpu import resolve_pyfock_gpu from .base import ( EngineCapabilities, EngineRequest, @@ -59,14 +60,15 @@ def capabilities(self) -> EngineCapabilities: supported_basis_sets=_PYFOCK_BASES, supports_solvent=False, supports_checkpoint_warm_start=False, - supports_gpu=False, + supports_gpu=True, supports_post_hf=False, supports_orbital_export=True, platform_notes=( "Neutral, closed-shell PBE single points and geometry optimizations. " "Density fitting and analytical gradients are used; hybrids, " - "solvent, checkpoints, and GPU are gated off. " - "Install with pip install quantui[pyfock]." + "solvent, and checkpoints are gated off. Optional PyFock GPU " + "acceleration uses CuPy; install with " + "pip install 'quantui[pyfock,pyfock-gpu-cuda12x]'." ), recommended_auxbasis=_AUX_BASIS, version=_pyfock_version(), @@ -87,6 +89,9 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: [symbol, float(x), float(y), float(z)] for symbol, (x, y, z) in zip(atoms, coordinates) ] + _use_gpu, _gpu_name, _gpu_reason = resolve_pyfock_gpu( + request.options.get("use_gpu") + ) try: with redirect_stdout(stream), redirect_stderr(stream): @@ -105,6 +110,10 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: f"Engine: PyFock {_pyfock_version() or 'unknown'} | " f"{request.method}/{request.basis} | density fitting: on" ) + if _use_gpu: + print(f"GPU acceleration: active ({_gpu_name}) — CuPy/Numba path") + elif request.options.get("use_gpu") is not False and _gpu_reason: + print(f"GPU acceleration: unavailable — {_gpu_reason}") mol = Mol(atoms=pyfock_atoms, charge=0) basis = Basis( mol, @@ -123,7 +132,7 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: request.options.get("conv_crit"), default=1.0e-7 ), ncores=_positive_int(request.options.get("ncores"), default=1), - use_gpu=False, + use_gpu=_use_gpu, ) dft.max_itr = _positive_int( request.options.get("max_iterations"), default=50 @@ -159,7 +168,16 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: getattr(dft, "mo_energies", None), getattr(dft, "mo_occupations", None), ) + _gpu_used = bool(getattr(dft, "use_gpu", False)) + if not _gpu_used: + _gpu_name = None warnings = ["PyFock uses density fitting with def2-universal-jfit."] + if ( + not _gpu_used + and request.options.get("use_gpu") is not False + and _gpu_reason + ): + warnings.append(f"PyFock GPU unavailable; CPU fallback: {_gpu_reason}") if not bool(getattr(dft, "converged", False)): warnings.append("PyFock reached its iteration limit without convergence.") @@ -175,6 +193,8 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: formula=_formula(atoms), homo_lumo_gap_ev=gap_ev, warnings=warnings, + gpu_used=_gpu_used, + gpu_name=_gpu_name, ) native = result.to_session_result() _attach_analysis(native, mol, basis, dft, density, atoms, coordinates, warnings) @@ -202,6 +222,7 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: expected_steps=request.options.get("expected_steps"), engine_id=self.engine_id, ncores=_positive_int(request.options.get("ncores"), default=1), + use_gpu=request.options.get("use_gpu"), ) return EngineResult( request_id=request.request_id, @@ -213,6 +234,8 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: method=native.method, basis=native.basis, formula=native.formula, + gpu_used=getattr(native, "gpu_used", False), + gpu_name=getattr(native, "gpu_name", None), native_result=native, ) diff --git a/quantui/engines/pyscf_engine.py b/quantui/engines/pyscf_engine.py index afc9694..7231214 100644 --- a/quantui/engines/pyscf_engine.py +++ b/quantui/engines/pyscf_engine.py @@ -112,6 +112,8 @@ def run(self, request: EngineRequest) -> EngineResult: basis=native.basis, formula=native.formula, homo_lumo_gap_ev=native.homo_lumo_gap_ev, + gpu_used=native.gpu_used, + gpu_name=native.gpu_name, native_result=native, ) @@ -137,6 +139,7 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: resume=bool(request.options.get("resume", False)), scf_rescue=bool(request.options.get("scf_rescue", True)), engine_id=self.engine_id, + use_gpu=request.options.get("use_gpu"), ) return EngineResult( request_id=request.request_id, @@ -148,6 +151,8 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: method=native.method, basis=native.basis, formula=native.formula, + gpu_used=getattr(native, "gpu_used", False), + gpu_name=getattr(native, "gpu_name", None), native_result=native, ) diff --git a/quantui/optimizer.py b/quantui/optimizer.py index 09d790a..0a9085f 100644 --- a/quantui/optimizer.py +++ b/quantui/optimizer.py @@ -98,6 +98,7 @@ def __init__( status_label: str = "Optimizing geometry", expected_steps=None, scf_rescue: bool = True, + use_gpu: Optional[bool] = None, **kwargs, ) -> None: super().__init__(**kwargs) @@ -109,6 +110,11 @@ def __init__( # rescue helper on non-convergence (default on; a batch caller # can disable it via CalculationRequest.options). self.scf_rescue = scf_rescue + # ``None`` follows the shared GPU preference; ``False`` is a + # per-calculation opt-out used by portable engine requests. + self.use_gpu = use_gpu + self.gpu_used = False + self.gpu_name: Optional[str] = None # Cooperative-cancel predicate; checked per step + wired into # the per-step SCF callback (the SCF runs silent here, so the # stream-based cancel can't see it). @@ -218,6 +224,23 @@ def calculate( mf, self._density_fit_used = _try_density_fit(mf) + # This is deliberately inside ``calculate``: ASE rebuilds the + # PySCF object for every geometry, so offload must be requested for + # every force evaluation rather than only for the first step. + if self.use_gpu is not False: + from .gpu_offload import try_to_gpu + + mf, _gpu_used, _gpu_name = try_to_gpu(mf, method_upper) + if _gpu_used: + self.gpu_used = True + self.gpu_name = _gpu_name + try: + self.progress_stream.write( + f"\n🚀 GPU offload active — running on {_gpu_name}\n" + ) + except Exception: # noqa: BLE001 — stream is user-owned + pass + mf.verbose = 0 mf.stdout = _sink @@ -323,6 +346,11 @@ class OptimizationResult: pyscf_mol_atom: Optional[Any] = None # atom list at final geometry (Angstrom) pyscf_mol_basis: Optional[str] = None density_fit: bool = False + # GPU provenance for the final optimization. For PySCF this is true when + # at least one SCF/gradient evaluation was migrated successfully; for + # PyFock it records the guarded CuPy request used by its calculator. + gpu_used: bool = False + gpu_name: Optional[str] = None # AUDIT F04 — mirrors SessionResult.dispersion_applied: None (method # doesn't use D3), True/False (does, and pyscf.dftd3 was/wasn't # importable during the optimization). @@ -471,6 +499,7 @@ def optimize_geometry( scf_rescue: bool = True, engine_id: str = "pyscf", ncores: int = 1, + use_gpu: Optional[bool] = None, ) -> OptimizationResult: """ Optimize a molecular geometry at the QM level using ASE-BFGS. @@ -513,6 +542,11 @@ def optimize_geometry( scf_rescue: Whether each step's SCF automatically retries through the shared rescue helper on non-convergence (M-SCF-ROBUST, see :mod:`quantui.scf_robust`). Default ``True``. + use_gpu: Per-calculation GPU preference. ``False`` disables GPU + migration; ``None`` follows the persistent QuantUI setting and + runtime probe. For PyFock geometry optimization, GPU execution + uses numerical forces because PyFock 0.1.7's analytical gradient + implementation is CPU-only. Returns: :class:`OptimizationResult` containing the optimized molecule, @@ -634,8 +668,15 @@ def optimize_geometry( status_label=status_label, expected_steps=expected_steps, scf_rescue=scf_rescue, + use_gpu=use_gpu, ) else: + from .pyfock_gpu import resolve_pyfock_gpu + + _pyfock_use_gpu, _pyfock_gpu_name, _pyfock_gpu_reason = resolve_pyfock_gpu( + use_gpu + ) + _pyfock_force_mode = "numerical" if _pyfock_use_gpu else "analytical" _pyfock_tmp = tempfile.TemporaryDirectory(prefix="quantui-pyfock-opt-") atoms.calc = _PyFockCalculator( basis=basis, @@ -643,24 +684,43 @@ def optimize_geometry( charge=molecule.charge, directory=_pyfock_tmp.name, convergence_check="error", - force_mode="analytical", + force_mode=_pyfock_force_mode, xc=method, isDF=True, sao=True, conv_crit=1.0e-7, max_itr=50, ncores=max(1, int(ncores)), - use_gpu=False, + use_gpu=_pyfock_use_gpu, ) # Keep the directory alive for every BFGS force evaluation and expose # the same metadata hooks consumed by the shared result builder. atoms.calc._quantui_tmpdir = _pyfock_tmp atoms.calc._density_fit_used = True + atoms.calc._gpu_used = _pyfock_use_gpu + atoms.calc._gpu_name = _pyfock_gpu_name if _pyfock_use_gpu else None + atoms.calc.gpu_used = _pyfock_use_gpu + atoms.calc.gpu_name = _pyfock_gpu_name if _pyfock_use_gpu else None atoms.calc.dispersion_applied = None - _write_stream( - _stream, - "\nPyFock geometry optimization: analytical density-fitted gradients.\n", - ) + if _pyfock_use_gpu: + _write_stream( + _stream, + f"\n🚀 PyFock GPU acceleration active — running on " + f"{_pyfock_gpu_name}. GPU force evaluation uses numerical " + "gradients because PyFock analytical GPU gradients are not " + "available in the installed release.\n", + ) + else: + _write_stream( + _stream, + "\nPyFock geometry optimization: analytical density-fitted gradients.\n", + ) + if _pyfock_gpu_reason and use_gpu is not False: + _write_stream( + _stream, + f"\n⚠ PyFock GPU unavailable — using CPU analytical gradients: " + f"{_pyfock_gpu_reason}\n", + ) # PySCF gradients (called by ASE-BFGS at every # step) emit fd-2 stderr from libcint / BLAS. Wrap the full BFGS run @@ -826,6 +886,8 @@ def _report_opt_fraction() -> None: _opt_dipole: Optional[float] = None _opt_dipole_vec: Optional[List[float]] = None _opt_density_fit = bool(getattr(atoms.calc, "_density_fit_used", False)) + _opt_gpu_used = bool(getattr(atoms.calc, "gpu_used", False)) + _opt_gpu_name = getattr(atoms.calc, "gpu_name", None) try: import numpy as _np_mo @@ -947,6 +1009,8 @@ def _report_opt_fraction() -> None: pyscf_mol_atom=_opt_mol_atom, pyscf_mol_basis=_opt_mol_basis, density_fit=_opt_density_fit, + gpu_used=_opt_gpu_used, + gpu_name=_opt_gpu_name, atom_symbols=_opt_atom_symbols, mulliken_charges=_opt_mulliken, dipole_moment_debye=_opt_dipole, diff --git a/quantui/pyfock_gpu.py b/quantui/pyfock_gpu.py new file mode 100644 index 0000000..58d0d49 --- /dev/null +++ b/quantui/pyfock_gpu.py @@ -0,0 +1,118 @@ +"""Guarded CuPy detection for PyFock GPU acceleration. + +PyFock's GPU path is independent of ``gpu4pyscf``: it uses CuPy directly +inside PyFock's Numba/CUDA kernels. Keep this probe separate from +``gpu_offload`` so installing one backend can never make the other backend +appear usable. + +The probe is deliberately conservative. A PyFock GPU run is enabled only +when CuPy imports and reports at least one usable CUDA device, and it still +honours QuantUI's persistent GPU preference and process-wide opt-out. +""" + +from __future__ import annotations + +import logging +import os +from functools import lru_cache +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +_REASON_OK = "" +_REASON_ENV_DISABLED = "QUANTUI_DISABLE_GPU is set in the environment" +_REASON_SETTINGS_DISABLED = ( + "GPU acceleration is switched off in QuantUI settings " + "(Status tab → Settings → GPU offload)" +) +_REASON_NOT_INSTALLED = ( + "CuPy is not installed — install the PyFock GPU extra matching your CUDA " + "version, e.g. pip install 'quantui[pyfock-gpu-cuda12x]'" +) +_REASON_NO_DEVICE = "CuPy reports 0 CUDA devices" + + +def _gpu_enabled_in_settings() -> bool: + """Read the shared persistent GPU preference without raising.""" + try: + from quantui.user_settings import UserSettings + + return bool(UserSettings.load().compute.gpu_enabled) + except Exception as exc: # noqa: BLE001 — settings never gate startup + logger.debug("could not read GPU preference, assuming enabled: %s", exc) + return True + + +def _device_name(cupy_module: Any) -> str: + properties = cupy_module.cuda.runtime.getDeviceProperties(0) + raw_name = properties.get("name", b"GPU") + if isinstance(raw_name, bytes): + return raw_name.decode("utf-8", errors="replace") + return str(raw_name) + + +@lru_cache(maxsize=1) +def _probe_pyfock_gpu() -> tuple[bool, Optional[str], str]: + """Return ``(available, device_name, reason)`` for the PyFock GPU path.""" + if os.environ.get("QUANTUI_DISABLE_GPU", "").strip() in ("1", "true", "True"): + return False, None, _REASON_ENV_DISABLED + if not _gpu_enabled_in_settings(): + return False, None, _REASON_SETTINGS_DISABLED + + try: + import cupy as cp + except ModuleNotFoundError: + return False, None, _REASON_NOT_INSTALLED + except ImportError as exc: + logger.warning("CuPy is installed but failed to import: %s", exc) + return False, None, f"CuPy is installed but failed to import: {exc}" + except Exception as exc: # noqa: BLE001 — driver failures become CPU fallback + logger.warning("CuPy import raised %s: %s", type(exc).__name__, exc) + return False, None, f"CuPy import raised {type(exc).__name__}: {exc}" + + try: + if int(cp.cuda.runtime.getDeviceCount()) < 1: + return False, None, _REASON_NO_DEVICE + return True, _device_name(cp), _REASON_OK + except Exception as exc: # noqa: BLE001 — driver failures become CPU fallback + logger.warning("CuPy device probe failed: %s", exc) + return False, None, f"CuPy device probe failed: {exc}" + + +def probe_pyfock_gpu() -> tuple[bool, Optional[str], str]: + """Return availability, device name, and an actionable failure reason.""" + return _probe_pyfock_gpu() + + +def resolve_pyfock_gpu( + enabled: Optional[bool] = None, +) -> tuple[bool, Optional[str], str]: + """Resolve whether a calculation should request PyFock GPU execution. + + ``enabled=False`` is a per-calculation hard opt-out. ``None`` follows the + persistent QuantUI preference. An explicit ``True`` still falls back to + CPU when CuPy or CUDA is unavailable; callers receive the reason so they + can show the user what happened. + """ + if enabled is False: + return False, None, "GPU acceleration disabled for this calculation" + return probe_pyfock_gpu() + + +def is_pyfock_gpu_available() -> tuple[bool, Optional[str]]: + """Compatibility-friendly two-value view of :func:`probe_pyfock_gpu`.""" + available, name, _reason = probe_pyfock_gpu() + return available, name + + +def clear_pyfock_gpu_probe_cache() -> None: + """Clear the process-lifetime probe, primarily for settings/tests.""" + _probe_pyfock_gpu.cache_clear() + + +__all__ = [ + "clear_pyfock_gpu_probe_cache", + "is_pyfock_gpu_available", + "probe_pyfock_gpu", + "resolve_pyfock_gpu", +] diff --git a/tests/test_cli.py b/tests/test_cli.py index 1289027..9bf46f8 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -230,6 +230,20 @@ def test_consumer_gpu_gets_fp64_advisory(self, monkeypatch, isolated_log_dir): assert "consumer" in err assert "SLOWER" in err + def test_pyfock_engine_probe_is_separate(self, monkeypatch, isolated_log_dir): + import quantui.pyfock_gpu as _pfg + + monkeypatch.setattr( + _pfg, + "probe_pyfock_gpu", + lambda: (True, "NVIDIA H200", ""), + ) + rc, out, err = _capture(["gpu", "check", "--engine", "pyfock"]) + assert rc == 0 + assert "PyFock GPU available" in out + assert "NVIDIA H200" in out + assert err == "" + class TestAnalyticsBuild: """`quantui analytics build` — wraps analytics.build_dashboard.""" diff --git a/tests/test_gpu_geometry_integration.py b/tests/test_gpu_geometry_integration.py new file mode 100644 index 0000000..5347237 --- /dev/null +++ b/tests/test_gpu_geometry_integration.py @@ -0,0 +1,76 @@ +"""Opt-in real NVIDIA tests for geometry-optimization offload. + +Normal CI does not provide a CUDA device. Set +``QUANTUI_RUN_GPU_INTEGRATION=1`` on a Linux/WSL host with gpu4pyscf and a +working NVIDIA driver to execute the real gate. +""" + +from __future__ import annotations + +import os + +import pytest + +from quantui.molecule import Molecule + + +@pytest.mark.gpu_integration +@pytest.mark.skipif( + os.environ.get("QUANTUI_RUN_GPU_INTEGRATION") != "1", + reason="set QUANTUI_RUN_GPU_INTEGRATION=1 for the real NVIDIA check", +) +def test_geometry_optimizer_offloads_each_pyscf_scf_step(): + from quantui.gpu_offload import probe_gpu + from quantui.optimizer import optimize_geometry + + available, name, reason = probe_gpu() + if not available: + pytest.fail(f"GPU integration requested but unavailable: {reason}") + + result = optimize_geometry( + Molecule( + ["H", "H"], + [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]], + ), + method="RHF", + basis="STO-3G", + fmax=0.05, + steps=1, + ) + + assert result.gpu_used is True + assert result.gpu_name == name + assert result.converged or result.n_steps == 1 + + +@pytest.mark.pyfock_gpu +@pytest.mark.skipif( + os.environ.get("QUANTUI_RUN_PYFOCK_GPU_INTEGRATION") != "1", + reason="set QUANTUI_RUN_PYFOCK_GPU_INTEGRATION=1 for the real PyFock GPU check", +) +def test_pyfock_gpu_water_single_point_is_real(): + from quantui.engines import EngineRequest, PyfockEngine + + result = PyfockEngine().run( + EngineRequest( + request_id="pyfock-gpu-water", + calc_type="single_point", + method="PBE", + basis="def2-SVP", + charge=0, + multiplicity=1, + molecule={ + "atoms": ["O", "H", "H"], + "coordinates": [ + [0.0, 0.0, 0.11779], + [0.0, 0.755453, -0.471161], + [0.0, -0.755453, -0.471161], + ], + }, + options={"use_gpu": True, "ncores": 2, "max_iterations": 50}, + ) + ) + assert result.status == "success", result.error + assert result.converged is True + assert result.native_result.gpu_used is True + assert result.native_result.gpu_name diff --git a/tests/test_optimizer.py b/tests/test_optimizer.py index 161aa0f..a40824f 100644 --- a/tests/test_optimizer.py +++ b/tests/test_optimizer.py @@ -18,6 +18,7 @@ """ import io +from unittest.mock import patch import pytest @@ -357,6 +358,28 @@ def test_method_basis_recorded(self): assert result.method == "RHF" assert result.basis == "STO-3G" + @pyscf_only + @pytest.mark.slow + def test_geometry_optimizer_records_gpu_migration(self): + """The optimizer must use the same migration seam as single points.""" + from quantui.optimizer import optimize_geometry + + with patch( + "quantui.gpu_offload.try_to_gpu", + side_effect=lambda mf, _method: (mf, True, "Test GPU"), + ) as migrate: + result = optimize_geometry( + _h2(0.60), + method="RHF", + basis="STO-3G", + fmax=1e-6, + steps=1, + ) + + assert migrate.call_count >= 1 + assert result.gpu_used is True + assert result.gpu_name == "Test GPU" + class TestOptimizeGeometryEnergyAndConvergence: """Energy sanity checks and convergence behaviour.""" diff --git a/tests/test_pyfock_engine.py b/tests/test_pyfock_engine.py index 8394576..b1073d5 100644 --- a/tests/test_pyfock_engine.py +++ b/tests/test_pyfock_engine.py @@ -68,6 +68,7 @@ def __init__(self, mol, basis, auxbasis, **kwargs): self.basis = basis self.auxbasis = auxbasis self.kwargs = kwargs + self.use_gpu = bool(kwargs.get("use_gpu", False)) self.converged = True self.niter = 7 self.mo_energies = [-0.8, -0.4, 0.1] @@ -90,6 +91,7 @@ def test_capabilities_include_phase2_analysis_and_geometry(self): caps = PyfockEngine().capabilities() assert caps.supported_calc_types == ("single_point", "geometry_opt") assert caps.supports_orbital_export is True + assert caps.supports_gpu is True def test_water_result_and_stdout_capture(self): stream = io.StringIO() @@ -113,6 +115,10 @@ def test_water_result_and_stdout_capture(self): "quantui.engines.pyfock_engine._load_pyfock_api", return_value=(_FakeMol, _FakeBasis, _FakeDFT), ), + patch( + "quantui.engines.pyfock_engine.resolve_pyfock_gpu", + return_value=(True, "Fake PyFock GPU", ""), + ), patch.dict(sys.modules, {"pyfock": fake_pyfock}), ): result = PyfockEngine().run(request) @@ -132,6 +138,11 @@ def test_water_result_and_stdout_capture(self): 0.5 * 2.541746473 ) assert _FakeDFT.last.sao is True + assert _FakeDFT.last.kwargs["use_gpu"] is True + assert result.gpu_used is True + assert result.gpu_name == "Fake PyFock GPU" + assert result.native_result.gpu_used is True + assert result.native_result.gpu_name == "Fake PyFock GPU" assert "fake PyFock SCF output" in stream.getvalue() assert "density fitting: on" in stream.getvalue() diff --git a/tests/test_pyfock_gpu.py b/tests/test_pyfock_gpu.py new file mode 100644 index 0000000..a13c195 --- /dev/null +++ b/tests/test_pyfock_gpu.py @@ -0,0 +1,60 @@ +"""Platform-independent tests for the guarded PyFock CuPy path.""" + +from __future__ import annotations + +import sys +from types import ModuleType, SimpleNamespace + +from quantui.pyfock_gpu import ( + clear_pyfock_gpu_probe_cache, + probe_pyfock_gpu, + resolve_pyfock_gpu, +) + + +def _fake_cupy(count=1, name=b"NVIDIA H200"): + runtime = SimpleNamespace( + getDeviceCount=lambda: count, + getDeviceProperties=lambda _index: {"name": name}, + ) + module = ModuleType("cupy") + module.cuda = SimpleNamespace(runtime=runtime) + return module + + +def test_probe_accepts_cupy_device_without_gpu4pyscf(monkeypatch): + monkeypatch.setitem(sys.modules, "cupy", _fake_cupy()) + clear_pyfock_gpu_probe_cache() + + assert probe_pyfock_gpu() == (True, "NVIDIA H200", "") + + +def test_probe_rejects_no_cuda_device(monkeypatch): + monkeypatch.setitem(sys.modules, "cupy", _fake_cupy(count=0)) + clear_pyfock_gpu_probe_cache() + + available, name, reason = probe_pyfock_gpu() + assert available is False + assert name is None + assert "0 CUDA devices" in reason + + +def test_process_opt_out_wins_over_cupy(monkeypatch): + monkeypatch.setitem(sys.modules, "cupy", _fake_cupy()) + monkeypatch.setenv("QUANTUI_DISABLE_GPU", "1") + clear_pyfock_gpu_probe_cache() + + assert probe_pyfock_gpu()[0] is False + assert "QUANTUI_DISABLE_GPU" in probe_pyfock_gpu()[2] + + +def test_per_calculation_opt_out_does_not_probe(monkeypatch): + def fail_probe(): + raise AssertionError("GPU probe should not run for explicit opt-out") + + monkeypatch.setattr("quantui.pyfock_gpu.probe_pyfock_gpu", fail_probe) + assert resolve_pyfock_gpu(False) == ( + False, + None, + "GPU acceleration disabled for this calculation", + ) From 2908d92fdeeb3df7b2561a6394a6490efdf40464 Mon Sep 17 00:00:00 2001 From: QuantUI Bot Date: Sat, 19 Sep 2026 04:29:00 +0000 Subject: [PATCH 2/2] fix(tests): stop a real os.environ leak that flaked PR #125's new GPU probe tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit tests/test_est_cross_device_probe.py::test_force_cpu_true_sets_disable_gpu_env calls quantui.benchmarks._calibration_worker in-process to check that force_cpu=True sets QUANTUI_DISABLE_GPU=1 before any PySCF import. The worker sets that var directly on the real os.environ (by design, so a freshly-imported gpu_offload module sees it) rather than through monkeypatch, so nothing reverted it once the test's spy short-circuited the rest of the worker body. Every later test in the same pytest-xdist worker process inherited a permanently GPU-disabled environment. That was latent and harmless until this PR's new tests/test_pyfock_gpu.py added two tests that assert a fake CuPy device is detected — they never touch QUANTUI_DISABLE_GPU themselves, so a leaked "1" from the earlier test silently changed their expected outcome. Only Python 3.10's CI job happened to schedule the leaking test and the new probe tests onto the same xdist worker, which is why 3.9/3.11 stayed green while 3.10 failed with "QUANTUI_DISABLE_GPU is set in the environment" instead of the fake device tuple. Fix: wrap the worker call in try/finally and pop the var afterward. Plain os.environ.pop, not a second monkeypatch.delenv — delenv would have recorded the leaked "1" as the value to restore at teardown, reintroducing the exact same leak (verified both ways in isolation before landing this). Also hardened the two new PyFock probe tests to delenv the var defensively on entry, so this class of cross-test pollution can't silently change their result again. Verified: ran the leaking test immediately followed by the new PyFock probe tests in a single serial process — reproduced the exact CI failure before this change, confirmed clean after. Full suite (pytest -m "not network", Python 3.10, matching CI) passes with no other regressions. Contributions: - Claude (Sonnet 5): code edits, review, and conceptual discussion - Jonathan Schultz: overall vision, planning, review, and orchestration Co-authored-by: Jonathan Schultz Co-authored-by: Claude Sonnet 5 --- tests/test_est_cross_device_probe.py | 44 ++++++++++++++++++++-------- tests/test_pyfock_gpu.py | 5 ++++ 2 files changed, 36 insertions(+), 13 deletions(-) diff --git a/tests/test_est_cross_device_probe.py b/tests/test_est_cross_device_probe.py index 9aa1912..3fabbc0 100644 --- a/tests/test_est_cross_device_probe.py +++ b/tests/test_est_cross_device_probe.py @@ -230,19 +230,37 @@ def put(self, item): log_path = tmp_path / "cal.log" log_path.write_text("") - _calibration_worker( - ["H", "H"], - [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]], - 0, - 1, - "RHF", - "STO-3G", - "single_point", - str(log_path), - q, - "test-cal-id", - True, # force_cpu - ) + try: + _calibration_worker( + ["H", "H"], + [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]], + 0, + 1, + "RHF", + "STO-3G", + "single_point", + str(log_path), + q, + "test-cal-id", + True, # force_cpu + ) + finally: + # The worker sets this directly on the real os.environ (by + # design — it must be visible to a freshly-imported gpu_offload + # module), not through monkeypatch, so nothing auto-reverts it + # for the rest of this xdist worker's test session. Without this + # cleanup, later tests in the same worker (e.g. + # tests/test_pyfock_gpu.py's cupy-probe tests) inherit a + # permanently "disabled" GPU env and fail nondeterministically + # depending on test distribution. + # + # Plain os.environ.pop, not monkeypatch.delenv: the var already + # exists at this point (the worker just set it), so a second + # monkeypatch.delenv call here would record ITS pre-call value + # ("1") as what to restore at test teardown — reintroducing the + # exact leak this is meant to fix. + os.environ.pop("QUANTUI_DISABLE_GPU", None) + assert captured_env.get("QUANTUI_DISABLE_GPU") == "1" def test_force_cpu_false_does_not_touch_env(self, monkeypatch, tmp_path): diff --git a/tests/test_pyfock_gpu.py b/tests/test_pyfock_gpu.py index a13c195..b1ea46f 100644 --- a/tests/test_pyfock_gpu.py +++ b/tests/test_pyfock_gpu.py @@ -23,6 +23,10 @@ def _fake_cupy(count=1, name=b"NVIDIA H200"): def test_probe_accepts_cupy_device_without_gpu4pyscf(monkeypatch): + # Other suites (e.g. test_est_cross_device_probe.py) exercise a worker + # path that sets this directly on the real os.environ; guard against + # inheriting that state from an earlier test in the same xdist worker. + monkeypatch.delenv("QUANTUI_DISABLE_GPU", raising=False) monkeypatch.setitem(sys.modules, "cupy", _fake_cupy()) clear_pyfock_gpu_probe_cache() @@ -30,6 +34,7 @@ def test_probe_accepts_cupy_device_without_gpu4pyscf(monkeypatch): def test_probe_rejects_no_cuda_device(monkeypatch): + monkeypatch.delenv("QUANTUI_DISABLE_GPU", raising=False) monkeypatch.setitem(sys.modules, "cupy", _fake_cupy(count=0)) clear_pyfock_gpu_probe_cache()