diff --git a/README.md b/README.md index 6782cb3..0edfaad 100644 --- a/README.md +++ b/README.md @@ -145,8 +145,10 @@ selected when PySCF is absent). The validated subset supports neutral, closed-shell PBE single points and geometry optimizations with def2-SVP or def2-TZVP. Density fitting is always enabled, analytical gradients drive optimization, and orbital, Mulliken, dipole, and cube analysis are retained. -Hybrids, charged/open-shell systems, solvent, checkpoint warm starts, and GPU -remain PySCF-only. +Hybrids, charged/open-shell systems, solvent, and checkpoint warm starts remain +PySCF-only. Optional PyFock GPU acceleration is available separately through +CuPy; geometry optimization uses numerical forces on GPU because PyFock 0.1.7 +does not yet provide analytical GPU gradients. PySCF does not install on Windows natively. For the complete feature set, the [`apptainer/quantui.def`](https://github.com/The-Schultz-Lab/QuantUI/blob/main/apptainer/quantui.def) container bundles @@ -240,6 +242,27 @@ and result cards will display the compute device. Whenever gpu4pyscf can't offload a particular call, QuantUI falls back to CPU automatically and the result card reflects which device ran. +### Optional: PyFock GPU acceleration + +PyFock uses its own CuPy/Numba CUDA implementation; it does not use +`gpu4pyscf`. Install the PyFock engine and the CUDA-suffixed CuPy extra that +matches the NVIDIA driver reported by `nvidia-smi`: + +```bash +# CUDA 13.x driver +pip install "quantui[pyfock,pyfock-gpu-cuda13x]" + +# CUDA 12.x driver +pip install "quantui[pyfock,pyfock-gpu-cuda12x]" +``` + +PyFock GPU use is guarded independently: QuantUI requires CuPy to import and +report a CUDA device, and still honors the Settings GPU toggle and +`QUANTUI_DISABLE_GPU=1`. If the probe fails, PyFock falls back to CPU and +reports the reason. GPU single points use PyFock's GPU SCF/integral/XC path; +GPU geometry optimizations use numerical finite-difference forces because the +installed PyFock release's analytical gradient implementation is CPU-only. + ### Optional: GFN-FF metal pre-optimization (xtb) The classical (MMFF/UFF) pre-optimizer relies on RDKit's organic valence diff --git a/docs/CLI.md b/docs/CLI.md index ad78caa..2d2ff82 100644 --- a/docs/CLI.md +++ b/docs/CLI.md @@ -177,7 +177,8 @@ quantui log tail -n 200 | grep -i error | tail -5 Probe whether QuantUI's GPU offload path is functional in the current environment. This is the canonical one-liner for verifying that `gpu4pyscf` + `cupy` are installed correctly and that -`is_gpu_available()` will return `True` when the app runs. +`is_gpu_available()` will return `True` when the PySCF app path runs. Use +`--engine pyfock` to probe PyFock's independent CuPy path. ### Flags @@ -189,6 +190,9 @@ None. # Is GPU offload working right now? quantui gpu check +# Check the separate PyFock/CuPy path +quantui gpu check --engine pyfock + # Use in a shell condition if quantui gpu check; then echo "GPU mode" diff --git a/pyproject.toml b/pyproject.toml index bb5d169..4d5eea7 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -190,6 +190,16 @@ gpu-cuda13x = [ "cutensor-cu13", ] +# PyFock GPU acceleration is a separate CuPy-only path. Do not install or +# probe gpu4pyscf for a PyFock calculation: the two engines have independent +# CUDA implementations and capability gates. +pyfock-gpu-cuda12x = [ + "cupy-cuda12x", +] +pyfock-gpu-cuda13x = [ + "cupy-cuda13x", +] + # Notebook smoke-test dependencies notebook = [ "nbmake>=1.4.0", @@ -329,6 +339,8 @@ markers = [ "network: marks tests that require network connectivity", "notebook: marks notebook smoke tests", "pyfock: marks real PyFock integration/parity tests (opt in with QUANTUI_RUN_PYFOCK_INTEGRATION=1)", + "gpu_integration: marks real NVIDIA GPU integration tests (opt in with QUANTUI_RUN_GPU_INTEGRATION=1)", + "pyfock_gpu: marks real PyFock CuPy GPU tests (opt in with QUANTUI_RUN_PYFOCK_GPU_INTEGRATION=1)", ] [tool.coverage.run] diff --git a/quantui/app.py b/quantui/app.py index f513eeb..13df928 100644 --- a/quantui/app.py +++ b/quantui/app.py @@ -5812,7 +5812,11 @@ def _run_required_final_single_point(target_mol, reason: str): "atoms": list(target_mol.atoms), "coordinates": [list(c) for c in target_mol.coordinates], }, - options={"verbose": 4, "scf_rescue": True}, + options={ + "verbose": 4, + "scf_rescue": True, + "use_gpu": bool(self._user_settings.compute.gpu_enabled), + }, progress_stream=log, # type: ignore[arg-type] solvent=_solvent, ), @@ -6007,6 +6011,7 @@ def _run_required_final_single_point(target_mol, reason: str): ), "resume": _resume, "scf_rescue": True, + "use_gpu": bool(self._user_settings.compute.gpu_enabled), }, progress_stream=log, # type: ignore[arg-type] checkpoint=_ckpt, @@ -6430,7 +6435,11 @@ def _run_required_final_single_point(target_mol, reason: str): "atoms": list(calc_mol.atoms), "coordinates": [list(c) for c in calc_mol.coordinates], }, - options={"verbose": 4, "scf_rescue": True}, + options={ + "verbose": 4, + "scf_rescue": True, + "use_gpu": bool(self._user_settings.compute.gpu_enabled), + }, progress_stream=log, # type: ignore[arg-type] solvent=_solvent, checkpoint=_ckpt, diff --git a/quantui/backends/worker_payload.py b/quantui/backends/worker_payload.py index 80e4b32..3ded223 100644 --- a/quantui/backends/worker_payload.py +++ b/quantui/backends/worker_payload.py @@ -116,6 +116,8 @@ def optimization_result_payload(result, *, trajectory_file: str) -> Dict[str, An "method": result.method, "basis": result.basis, "formula": result.formula, + "gpu_used": bool(getattr(result, "gpu_used", False)), + "gpu_name": getattr(result, "gpu_name", None), "trajectory_file": trajectory_file, } diff --git a/quantui/cli.py b/quantui/cli.py index 6b27ba3..a9dffad 100644 --- a/quantui/cli.py +++ b/quantui/cli.py @@ -9,9 +9,9 @@ * ``quantui log tail [-n N]`` — print the last N event-log entries (default 20). Reads ``~/.quantui/logs/event_log.jsonl`` honoring the ``QUANTUI_LOG_DIR`` env override. -* ``quantui gpu check`` — run QuantUI's GPU-offload detection and print - ``(available, device-name)``. Exit code 0 when GPU is usable, 1 when - not — handy for one-line CI / shell-script gating. +* ``quantui gpu check [--engine pyscf|pyfock]`` — run one engine's GPU + detection and print ``(available, device-name)``. Exit code 0 when GPU is + usable, 1 when not — handy for one-line CI / shell-script gating. * ``quantui analytics build [-o PATH] [--open]`` — build a self-contained HTML analytics dashboard from ``perf_log.jsonl``. Default output: ``~/.quantui/dashboard.html``. Pass ``--open`` to automatically open @@ -106,19 +106,36 @@ def _cmd_gpu_check(args: argparse.Namespace) -> int: Returns exit code 0 when GPU offload is available, 1 when it's not — so ``if quantui gpu check; then ...; fi`` works in shell scripts. """ - from quantui.gpu_offload import is_gpu_available, is_low_fp64_device, probe_gpu - - # The detection probe is cached; clear so each CLI invocation is - # fresh (the user may have just installed gpu4pyscf and wants to - # confirm without restarting their shell). - # cache_clear is forwarded from _probe_gpu's lru_cache onto this function - # at definition time (gpu_offload.py); mypy can't see a monkey-patched - # attribute across the module boundary. - is_gpu_available.cache_clear() # type: ignore[attr-defined] - available, name, reason = probe_gpu() + engine = getattr(args, "engine", "pyscf") + if engine == "pyfock": + from quantui.pyfock_gpu import ( + clear_pyfock_gpu_probe_cache, + probe_pyfock_gpu, + ) + + clear_pyfock_gpu_probe_cache() + available, name, reason = probe_pyfock_gpu() + label = "PyFock GPU" + low_fp64 = False + else: + from quantui.gpu_offload import ( + is_gpu_available, + is_low_fp64_device, + probe_gpu, + ) + + # The detection probe is cached; clear so each CLI invocation is + # fresh (the user may have just installed gpu4pyscf and wants to + # confirm without restarting their shell). + # cache_clear is forwarded from _probe_gpu's lru_cache onto this + # function at definition time; mypy can't see that attribute. + is_gpu_available.cache_clear() # type: ignore[attr-defined] + available, name, reason = probe_gpu() + label = "GPU offload" + low_fp64 = is_low_fp64_device(name) if available: - print(f"GPU offload available: {name}") - if is_low_fp64_device(name): + print(f"{label} available: {name}") + if low_fp64: # Available is not the same as worth using: PySCF is FP64 # throughout, and consumer cards gate double precision to a small # fraction of single. Say so here rather than let the user discover @@ -391,7 +408,13 @@ def _build_parser() -> argparse.ArgumentParser: gpu_sub = gpu_parser.add_subparsers(dest="gpu_command", required=True) gpu_check = gpu_sub.add_parser( "check", - help="Run QuantUI's GPU-offload detection probe.", + help="Run a GPU detection probe.", + ) + gpu_check.add_argument( + "--engine", + choices=("pyscf", "pyfock"), + default="pyscf", + help="Engine-specific GPU path to probe (default: pyscf).", ) gpu_check.set_defaults(func=_cmd_gpu_check) diff --git a/quantui/engines/base.py b/quantui/engines/base.py index 8ddd8bb..b53d2b6 100644 --- a/quantui/engines/base.py +++ b/quantui/engines/base.py @@ -1,7 +1,8 @@ """ Quantum-engine contract types (v0.1). -See ``QuantUI-development-tracking/TODO/QUANTUM-ENGINE-CONTRACT.md``. +The engine contract is maintained alongside the project's development +documentation. """ from __future__ import annotations @@ -85,6 +86,8 @@ class EngineResult: homo_lumo_gap_ev: Optional[float] = None warnings: List[str] = field(default_factory=list) error: Optional[Dict[str, Any]] = None + gpu_used: bool = False + gpu_name: Optional[str] = None native_result: Optional[Any] = field(default=None, repr=False, compare=False) def to_dict(self) -> Dict[str, Any]: @@ -118,6 +121,8 @@ def to_session_result(self) -> Any: basis=self.basis, formula=self.formula, density_fit=self.engine_id == "pyfock", + gpu_used=self.gpu_used, + gpu_name=self.gpu_name, scf_variant="RKS" if self.engine_id == "pyfock" else "", engine_id=self.engine_id, ) diff --git a/quantui/engines/pyfock_engine.py b/quantui/engines/pyfock_engine.py index c683dcc..69fb16d 100644 --- a/quantui/engines/pyfock_engine.py +++ b/quantui/engines/pyfock_engine.py @@ -7,6 +7,7 @@ from contextlib import redirect_stderr, redirect_stdout from typing import Any, Optional +from ..pyfock_gpu import resolve_pyfock_gpu from .base import ( EngineCapabilities, EngineRequest, @@ -59,14 +60,15 @@ def capabilities(self) -> EngineCapabilities: supported_basis_sets=_PYFOCK_BASES, supports_solvent=False, supports_checkpoint_warm_start=False, - supports_gpu=False, + supports_gpu=True, supports_post_hf=False, supports_orbital_export=True, platform_notes=( "Neutral, closed-shell PBE single points and geometry optimizations. " "Density fitting and analytical gradients are used; hybrids, " - "solvent, checkpoints, and GPU are gated off. " - "Install with pip install quantui[pyfock]." + "solvent, and checkpoints are gated off. Optional PyFock GPU " + "acceleration uses CuPy; install with " + "pip install 'quantui[pyfock,pyfock-gpu-cuda12x]'." ), recommended_auxbasis=_AUX_BASIS, version=_pyfock_version(), @@ -87,6 +89,9 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: [symbol, float(x), float(y), float(z)] for symbol, (x, y, z) in zip(atoms, coordinates) ] + _use_gpu, _gpu_name, _gpu_reason = resolve_pyfock_gpu( + request.options.get("use_gpu") + ) try: with redirect_stdout(stream), redirect_stderr(stream): @@ -105,6 +110,10 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: f"Engine: PyFock {_pyfock_version() or 'unknown'} | " f"{request.method}/{request.basis} | density fitting: on" ) + if _use_gpu: + print(f"GPU acceleration: active ({_gpu_name}) — CuPy/Numba path") + elif request.options.get("use_gpu") is not False and _gpu_reason: + print(f"GPU acceleration: unavailable — {_gpu_reason}") mol = Mol(atoms=pyfock_atoms, charge=0) basis = Basis( mol, @@ -123,7 +132,7 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: request.options.get("conv_crit"), default=1.0e-7 ), ncores=_positive_int(request.options.get("ncores"), default=1), - use_gpu=False, + use_gpu=_use_gpu, ) dft.max_itr = _positive_int( request.options.get("max_iterations"), default=50 @@ -159,7 +168,16 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: getattr(dft, "mo_energies", None), getattr(dft, "mo_occupations", None), ) + _gpu_used = bool(getattr(dft, "use_gpu", False)) + if not _gpu_used: + _gpu_name = None warnings = ["PyFock uses density fitting with def2-universal-jfit."] + if ( + not _gpu_used + and request.options.get("use_gpu") is not False + and _gpu_reason + ): + warnings.append(f"PyFock GPU unavailable; CPU fallback: {_gpu_reason}") if not bool(getattr(dft, "converged", False)): warnings.append("PyFock reached its iteration limit without convergence.") @@ -175,6 +193,8 @@ def _run_single_point(self, request: EngineRequest) -> EngineResult: formula=_formula(atoms), homo_lumo_gap_ev=gap_ev, warnings=warnings, + gpu_used=_gpu_used, + gpu_name=_gpu_name, ) native = result.to_session_result() _attach_analysis(native, mol, basis, dft, density, atoms, coordinates, warnings) @@ -202,6 +222,7 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: expected_steps=request.options.get("expected_steps"), engine_id=self.engine_id, ncores=_positive_int(request.options.get("ncores"), default=1), + use_gpu=request.options.get("use_gpu"), ) return EngineResult( request_id=request.request_id, @@ -213,6 +234,8 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: method=native.method, basis=native.basis, formula=native.formula, + gpu_used=getattr(native, "gpu_used", False), + gpu_name=getattr(native, "gpu_name", None), native_result=native, ) diff --git a/quantui/engines/pyscf_engine.py b/quantui/engines/pyscf_engine.py index afc9694..7231214 100644 --- a/quantui/engines/pyscf_engine.py +++ b/quantui/engines/pyscf_engine.py @@ -112,6 +112,8 @@ def run(self, request: EngineRequest) -> EngineResult: basis=native.basis, formula=native.formula, homo_lumo_gap_ev=native.homo_lumo_gap_ev, + gpu_used=native.gpu_used, + gpu_name=native.gpu_name, native_result=native, ) @@ -137,6 +139,7 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: resume=bool(request.options.get("resume", False)), scf_rescue=bool(request.options.get("scf_rescue", True)), engine_id=self.engine_id, + use_gpu=request.options.get("use_gpu"), ) return EngineResult( request_id=request.request_id, @@ -148,6 +151,8 @@ def _run_geometry_opt(self, request: EngineRequest) -> EngineResult: method=native.method, basis=native.basis, formula=native.formula, + gpu_used=getattr(native, "gpu_used", False), + gpu_name=getattr(native, "gpu_name", None), native_result=native, ) diff --git a/quantui/optimizer.py b/quantui/optimizer.py index 09d790a..0a9085f 100644 --- a/quantui/optimizer.py +++ b/quantui/optimizer.py @@ -98,6 +98,7 @@ def __init__( status_label: str = "Optimizing geometry", expected_steps=None, scf_rescue: bool = True, + use_gpu: Optional[bool] = None, **kwargs, ) -> None: super().__init__(**kwargs) @@ -109,6 +110,11 @@ def __init__( # rescue helper on non-convergence (default on; a batch caller # can disable it via CalculationRequest.options). self.scf_rescue = scf_rescue + # ``None`` follows the shared GPU preference; ``False`` is a + # per-calculation opt-out used by portable engine requests. + self.use_gpu = use_gpu + self.gpu_used = False + self.gpu_name: Optional[str] = None # Cooperative-cancel predicate; checked per step + wired into # the per-step SCF callback (the SCF runs silent here, so the # stream-based cancel can't see it). @@ -218,6 +224,23 @@ def calculate( mf, self._density_fit_used = _try_density_fit(mf) + # This is deliberately inside ``calculate``: ASE rebuilds the + # PySCF object for every geometry, so offload must be requested for + # every force evaluation rather than only for the first step. + if self.use_gpu is not False: + from .gpu_offload import try_to_gpu + + mf, _gpu_used, _gpu_name = try_to_gpu(mf, method_upper) + if _gpu_used: + self.gpu_used = True + self.gpu_name = _gpu_name + try: + self.progress_stream.write( + f"\n🚀 GPU offload active — running on {_gpu_name}\n" + ) + except Exception: # noqa: BLE001 — stream is user-owned + pass + mf.verbose = 0 mf.stdout = _sink @@ -323,6 +346,11 @@ class OptimizationResult: pyscf_mol_atom: Optional[Any] = None # atom list at final geometry (Angstrom) pyscf_mol_basis: Optional[str] = None density_fit: bool = False + # GPU provenance for the final optimization. For PySCF this is true when + # at least one SCF/gradient evaluation was migrated successfully; for + # PyFock it records the guarded CuPy request used by its calculator. + gpu_used: bool = False + gpu_name: Optional[str] = None # AUDIT F04 — mirrors SessionResult.dispersion_applied: None (method # doesn't use D3), True/False (does, and pyscf.dftd3 was/wasn't # importable during the optimization). @@ -471,6 +499,7 @@ def optimize_geometry( scf_rescue: bool = True, engine_id: str = "pyscf", ncores: int = 1, + use_gpu: Optional[bool] = None, ) -> OptimizationResult: """ Optimize a molecular geometry at the QM level using ASE-BFGS. @@ -513,6 +542,11 @@ def optimize_geometry( scf_rescue: Whether each step's SCF automatically retries through the shared rescue helper on non-convergence (M-SCF-ROBUST, see :mod:`quantui.scf_robust`). Default ``True``. + use_gpu: Per-calculation GPU preference. ``False`` disables GPU + migration; ``None`` follows the persistent QuantUI setting and + runtime probe. For PyFock geometry optimization, GPU execution + uses numerical forces because PyFock 0.1.7's analytical gradient + implementation is CPU-only. Returns: :class:`OptimizationResult` containing the optimized molecule, @@ -634,8 +668,15 @@ def optimize_geometry( status_label=status_label, expected_steps=expected_steps, scf_rescue=scf_rescue, + use_gpu=use_gpu, ) else: + from .pyfock_gpu import resolve_pyfock_gpu + + _pyfock_use_gpu, _pyfock_gpu_name, _pyfock_gpu_reason = resolve_pyfock_gpu( + use_gpu + ) + _pyfock_force_mode = "numerical" if _pyfock_use_gpu else "analytical" _pyfock_tmp = tempfile.TemporaryDirectory(prefix="quantui-pyfock-opt-") atoms.calc = _PyFockCalculator( basis=basis, @@ -643,24 +684,43 @@ def optimize_geometry( charge=molecule.charge, directory=_pyfock_tmp.name, convergence_check="error", - force_mode="analytical", + force_mode=_pyfock_force_mode, xc=method, isDF=True, sao=True, conv_crit=1.0e-7, max_itr=50, ncores=max(1, int(ncores)), - use_gpu=False, + use_gpu=_pyfock_use_gpu, ) # Keep the directory alive for every BFGS force evaluation and expose # the same metadata hooks consumed by the shared result builder. atoms.calc._quantui_tmpdir = _pyfock_tmp atoms.calc._density_fit_used = True + atoms.calc._gpu_used = _pyfock_use_gpu + atoms.calc._gpu_name = _pyfock_gpu_name if _pyfock_use_gpu else None + atoms.calc.gpu_used = _pyfock_use_gpu + atoms.calc.gpu_name = _pyfock_gpu_name if _pyfock_use_gpu else None atoms.calc.dispersion_applied = None - _write_stream( - _stream, - "\nPyFock geometry optimization: analytical density-fitted gradients.\n", - ) + if _pyfock_use_gpu: + _write_stream( + _stream, + f"\n🚀 PyFock GPU acceleration active — running on " + f"{_pyfock_gpu_name}. GPU force evaluation uses numerical " + "gradients because PyFock analytical GPU gradients are not " + "available in the installed release.\n", + ) + else: + _write_stream( + _stream, + "\nPyFock geometry optimization: analytical density-fitted gradients.\n", + ) + if _pyfock_gpu_reason and use_gpu is not False: + _write_stream( + _stream, + f"\n⚠ PyFock GPU unavailable — using CPU analytical gradients: " + f"{_pyfock_gpu_reason}\n", + ) # PySCF gradients (called by ASE-BFGS at every # step) emit fd-2 stderr from libcint / BLAS. Wrap the full BFGS run @@ -826,6 +886,8 @@ def _report_opt_fraction() -> None: _opt_dipole: Optional[float] = None _opt_dipole_vec: Optional[List[float]] = None _opt_density_fit = bool(getattr(atoms.calc, "_density_fit_used", False)) + _opt_gpu_used = bool(getattr(atoms.calc, "gpu_used", False)) + _opt_gpu_name = getattr(atoms.calc, "gpu_name", None) try: import numpy as _np_mo @@ -947,6 +1009,8 @@ def _report_opt_fraction() -> None: pyscf_mol_atom=_opt_mol_atom, pyscf_mol_basis=_opt_mol_basis, density_fit=_opt_density_fit, + gpu_used=_opt_gpu_used, + gpu_name=_opt_gpu_name, atom_symbols=_opt_atom_symbols, mulliken_charges=_opt_mulliken, dipole_moment_debye=_opt_dipole, diff --git a/quantui/pyfock_gpu.py b/quantui/pyfock_gpu.py new file mode 100644 index 0000000..58d0d49 --- /dev/null +++ b/quantui/pyfock_gpu.py @@ -0,0 +1,118 @@ +"""Guarded CuPy detection for PyFock GPU acceleration. + +PyFock's GPU path is independent of ``gpu4pyscf``: it uses CuPy directly +inside PyFock's Numba/CUDA kernels. Keep this probe separate from +``gpu_offload`` so installing one backend can never make the other backend +appear usable. + +The probe is deliberately conservative. A PyFock GPU run is enabled only +when CuPy imports and reports at least one usable CUDA device, and it still +honours QuantUI's persistent GPU preference and process-wide opt-out. +""" + +from __future__ import annotations + +import logging +import os +from functools import lru_cache +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +_REASON_OK = "" +_REASON_ENV_DISABLED = "QUANTUI_DISABLE_GPU is set in the environment" +_REASON_SETTINGS_DISABLED = ( + "GPU acceleration is switched off in QuantUI settings " + "(Status tab → Settings → GPU offload)" +) +_REASON_NOT_INSTALLED = ( + "CuPy is not installed — install the PyFock GPU extra matching your CUDA " + "version, e.g. pip install 'quantui[pyfock-gpu-cuda12x]'" +) +_REASON_NO_DEVICE = "CuPy reports 0 CUDA devices" + + +def _gpu_enabled_in_settings() -> bool: + """Read the shared persistent GPU preference without raising.""" + try: + from quantui.user_settings import UserSettings + + return bool(UserSettings.load().compute.gpu_enabled) + except Exception as exc: # noqa: BLE001 — settings never gate startup + logger.debug("could not read GPU preference, assuming enabled: %s", exc) + return True + + +def _device_name(cupy_module: Any) -> str: + properties = cupy_module.cuda.runtime.getDeviceProperties(0) + raw_name = properties.get("name", b"GPU") + if isinstance(raw_name, bytes): + return raw_name.decode("utf-8", errors="replace") + return str(raw_name) + + +@lru_cache(maxsize=1) +def _probe_pyfock_gpu() -> tuple[bool, Optional[str], str]: + """Return ``(available, device_name, reason)`` for the PyFock GPU path.""" + if os.environ.get("QUANTUI_DISABLE_GPU", "").strip() in ("1", "true", "True"): + return False, None, _REASON_ENV_DISABLED + if not _gpu_enabled_in_settings(): + return False, None, _REASON_SETTINGS_DISABLED + + try: + import cupy as cp + except ModuleNotFoundError: + return False, None, _REASON_NOT_INSTALLED + except ImportError as exc: + logger.warning("CuPy is installed but failed to import: %s", exc) + return False, None, f"CuPy is installed but failed to import: {exc}" + except Exception as exc: # noqa: BLE001 — driver failures become CPU fallback + logger.warning("CuPy import raised %s: %s", type(exc).__name__, exc) + return False, None, f"CuPy import raised {type(exc).__name__}: {exc}" + + try: + if int(cp.cuda.runtime.getDeviceCount()) < 1: + return False, None, _REASON_NO_DEVICE + return True, _device_name(cp), _REASON_OK + except Exception as exc: # noqa: BLE001 — driver failures become CPU fallback + logger.warning("CuPy device probe failed: %s", exc) + return False, None, f"CuPy device probe failed: {exc}" + + +def probe_pyfock_gpu() -> tuple[bool, Optional[str], str]: + """Return availability, device name, and an actionable failure reason.""" + return _probe_pyfock_gpu() + + +def resolve_pyfock_gpu( + enabled: Optional[bool] = None, +) -> tuple[bool, Optional[str], str]: + """Resolve whether a calculation should request PyFock GPU execution. + + ``enabled=False`` is a per-calculation hard opt-out. ``None`` follows the + persistent QuantUI preference. An explicit ``True`` still falls back to + CPU when CuPy or CUDA is unavailable; callers receive the reason so they + can show the user what happened. + """ + if enabled is False: + return False, None, "GPU acceleration disabled for this calculation" + return probe_pyfock_gpu() + + +def is_pyfock_gpu_available() -> tuple[bool, Optional[str]]: + """Compatibility-friendly two-value view of :func:`probe_pyfock_gpu`.""" + available, name, _reason = probe_pyfock_gpu() + return available, name + + +def clear_pyfock_gpu_probe_cache() -> None: + """Clear the process-lifetime probe, primarily for settings/tests.""" + _probe_pyfock_gpu.cache_clear() + + +__all__ = [ + "clear_pyfock_gpu_probe_cache", + "is_pyfock_gpu_available", + "probe_pyfock_gpu", + "resolve_pyfock_gpu", +] diff --git a/tests/test_cli.py b/tests/test_cli.py index 1289027..9bf46f8 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -230,6 +230,20 @@ def test_consumer_gpu_gets_fp64_advisory(self, monkeypatch, isolated_log_dir): assert "consumer" in err assert "SLOWER" in err + def test_pyfock_engine_probe_is_separate(self, monkeypatch, isolated_log_dir): + import quantui.pyfock_gpu as _pfg + + monkeypatch.setattr( + _pfg, + "probe_pyfock_gpu", + lambda: (True, "NVIDIA H200", ""), + ) + rc, out, err = _capture(["gpu", "check", "--engine", "pyfock"]) + assert rc == 0 + assert "PyFock GPU available" in out + assert "NVIDIA H200" in out + assert err == "" + class TestAnalyticsBuild: """`quantui analytics build` — wraps analytics.build_dashboard.""" diff --git a/tests/test_est_cross_device_probe.py b/tests/test_est_cross_device_probe.py index 9aa1912..3fabbc0 100644 --- a/tests/test_est_cross_device_probe.py +++ b/tests/test_est_cross_device_probe.py @@ -230,19 +230,37 @@ def put(self, item): log_path = tmp_path / "cal.log" log_path.write_text("") - _calibration_worker( - ["H", "H"], - [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]], - 0, - 1, - "RHF", - "STO-3G", - "single_point", - str(log_path), - q, - "test-cal-id", - True, # force_cpu - ) + try: + _calibration_worker( + ["H", "H"], + [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]], + 0, + 1, + "RHF", + "STO-3G", + "single_point", + str(log_path), + q, + "test-cal-id", + True, # force_cpu + ) + finally: + # The worker sets this directly on the real os.environ (by + # design — it must be visible to a freshly-imported gpu_offload + # module), not through monkeypatch, so nothing auto-reverts it + # for the rest of this xdist worker's test session. Without this + # cleanup, later tests in the same worker (e.g. + # tests/test_pyfock_gpu.py's cupy-probe tests) inherit a + # permanently "disabled" GPU env and fail nondeterministically + # depending on test distribution. + # + # Plain os.environ.pop, not monkeypatch.delenv: the var already + # exists at this point (the worker just set it), so a second + # monkeypatch.delenv call here would record ITS pre-call value + # ("1") as what to restore at test teardown — reintroducing the + # exact leak this is meant to fix. + os.environ.pop("QUANTUI_DISABLE_GPU", None) + assert captured_env.get("QUANTUI_DISABLE_GPU") == "1" def test_force_cpu_false_does_not_touch_env(self, monkeypatch, tmp_path): diff --git a/tests/test_gpu_geometry_integration.py b/tests/test_gpu_geometry_integration.py new file mode 100644 index 0000000..5347237 --- /dev/null +++ b/tests/test_gpu_geometry_integration.py @@ -0,0 +1,76 @@ +"""Opt-in real NVIDIA tests for geometry-optimization offload. + +Normal CI does not provide a CUDA device. Set +``QUANTUI_RUN_GPU_INTEGRATION=1`` on a Linux/WSL host with gpu4pyscf and a +working NVIDIA driver to execute the real gate. +""" + +from __future__ import annotations + +import os + +import pytest + +from quantui.molecule import Molecule + + +@pytest.mark.gpu_integration +@pytest.mark.skipif( + os.environ.get("QUANTUI_RUN_GPU_INTEGRATION") != "1", + reason="set QUANTUI_RUN_GPU_INTEGRATION=1 for the real NVIDIA check", +) +def test_geometry_optimizer_offloads_each_pyscf_scf_step(): + from quantui.gpu_offload import probe_gpu + from quantui.optimizer import optimize_geometry + + available, name, reason = probe_gpu() + if not available: + pytest.fail(f"GPU integration requested but unavailable: {reason}") + + result = optimize_geometry( + Molecule( + ["H", "H"], + [[0.0, 0.0, 0.0], [0.0, 0.0, 0.74]], + ), + method="RHF", + basis="STO-3G", + fmax=0.05, + steps=1, + ) + + assert result.gpu_used is True + assert result.gpu_name == name + assert result.converged or result.n_steps == 1 + + +@pytest.mark.pyfock_gpu +@pytest.mark.skipif( + os.environ.get("QUANTUI_RUN_PYFOCK_GPU_INTEGRATION") != "1", + reason="set QUANTUI_RUN_PYFOCK_GPU_INTEGRATION=1 for the real PyFock GPU check", +) +def test_pyfock_gpu_water_single_point_is_real(): + from quantui.engines import EngineRequest, PyfockEngine + + result = PyfockEngine().run( + EngineRequest( + request_id="pyfock-gpu-water", + calc_type="single_point", + method="PBE", + basis="def2-SVP", + charge=0, + multiplicity=1, + molecule={ + "atoms": ["O", "H", "H"], + "coordinates": [ + [0.0, 0.0, 0.11779], + [0.0, 0.755453, -0.471161], + [0.0, -0.755453, -0.471161], + ], + }, + options={"use_gpu": True, "ncores": 2, "max_iterations": 50}, + ) + ) + assert result.status == "success", result.error + assert result.converged is True + assert result.native_result.gpu_used is True + assert result.native_result.gpu_name diff --git a/tests/test_optimizer.py b/tests/test_optimizer.py index 161aa0f..a40824f 100644 --- a/tests/test_optimizer.py +++ b/tests/test_optimizer.py @@ -18,6 +18,7 @@ """ import io +from unittest.mock import patch import pytest @@ -357,6 +358,28 @@ def test_method_basis_recorded(self): assert result.method == "RHF" assert result.basis == "STO-3G" + @pyscf_only + @pytest.mark.slow + def test_geometry_optimizer_records_gpu_migration(self): + """The optimizer must use the same migration seam as single points.""" + from quantui.optimizer import optimize_geometry + + with patch( + "quantui.gpu_offload.try_to_gpu", + side_effect=lambda mf, _method: (mf, True, "Test GPU"), + ) as migrate: + result = optimize_geometry( + _h2(0.60), + method="RHF", + basis="STO-3G", + fmax=1e-6, + steps=1, + ) + + assert migrate.call_count >= 1 + assert result.gpu_used is True + assert result.gpu_name == "Test GPU" + class TestOptimizeGeometryEnergyAndConvergence: """Energy sanity checks and convergence behaviour.""" diff --git a/tests/test_pyfock_engine.py b/tests/test_pyfock_engine.py index 8394576..b1073d5 100644 --- a/tests/test_pyfock_engine.py +++ b/tests/test_pyfock_engine.py @@ -68,6 +68,7 @@ def __init__(self, mol, basis, auxbasis, **kwargs): self.basis = basis self.auxbasis = auxbasis self.kwargs = kwargs + self.use_gpu = bool(kwargs.get("use_gpu", False)) self.converged = True self.niter = 7 self.mo_energies = [-0.8, -0.4, 0.1] @@ -90,6 +91,7 @@ def test_capabilities_include_phase2_analysis_and_geometry(self): caps = PyfockEngine().capabilities() assert caps.supported_calc_types == ("single_point", "geometry_opt") assert caps.supports_orbital_export is True + assert caps.supports_gpu is True def test_water_result_and_stdout_capture(self): stream = io.StringIO() @@ -113,6 +115,10 @@ def test_water_result_and_stdout_capture(self): "quantui.engines.pyfock_engine._load_pyfock_api", return_value=(_FakeMol, _FakeBasis, _FakeDFT), ), + patch( + "quantui.engines.pyfock_engine.resolve_pyfock_gpu", + return_value=(True, "Fake PyFock GPU", ""), + ), patch.dict(sys.modules, {"pyfock": fake_pyfock}), ): result = PyfockEngine().run(request) @@ -132,6 +138,11 @@ def test_water_result_and_stdout_capture(self): 0.5 * 2.541746473 ) assert _FakeDFT.last.sao is True + assert _FakeDFT.last.kwargs["use_gpu"] is True + assert result.gpu_used is True + assert result.gpu_name == "Fake PyFock GPU" + assert result.native_result.gpu_used is True + assert result.native_result.gpu_name == "Fake PyFock GPU" assert "fake PyFock SCF output" in stream.getvalue() assert "density fitting: on" in stream.getvalue() diff --git a/tests/test_pyfock_gpu.py b/tests/test_pyfock_gpu.py new file mode 100644 index 0000000..b1ea46f --- /dev/null +++ b/tests/test_pyfock_gpu.py @@ -0,0 +1,65 @@ +"""Platform-independent tests for the guarded PyFock CuPy path.""" + +from __future__ import annotations + +import sys +from types import ModuleType, SimpleNamespace + +from quantui.pyfock_gpu import ( + clear_pyfock_gpu_probe_cache, + probe_pyfock_gpu, + resolve_pyfock_gpu, +) + + +def _fake_cupy(count=1, name=b"NVIDIA H200"): + runtime = SimpleNamespace( + getDeviceCount=lambda: count, + getDeviceProperties=lambda _index: {"name": name}, + ) + module = ModuleType("cupy") + module.cuda = SimpleNamespace(runtime=runtime) + return module + + +def test_probe_accepts_cupy_device_without_gpu4pyscf(monkeypatch): + # Other suites (e.g. test_est_cross_device_probe.py) exercise a worker + # path that sets this directly on the real os.environ; guard against + # inheriting that state from an earlier test in the same xdist worker. + monkeypatch.delenv("QUANTUI_DISABLE_GPU", raising=False) + monkeypatch.setitem(sys.modules, "cupy", _fake_cupy()) + clear_pyfock_gpu_probe_cache() + + assert probe_pyfock_gpu() == (True, "NVIDIA H200", "") + + +def test_probe_rejects_no_cuda_device(monkeypatch): + monkeypatch.delenv("QUANTUI_DISABLE_GPU", raising=False) + monkeypatch.setitem(sys.modules, "cupy", _fake_cupy(count=0)) + clear_pyfock_gpu_probe_cache() + + available, name, reason = probe_pyfock_gpu() + assert available is False + assert name is None + assert "0 CUDA devices" in reason + + +def test_process_opt_out_wins_over_cupy(monkeypatch): + monkeypatch.setitem(sys.modules, "cupy", _fake_cupy()) + monkeypatch.setenv("QUANTUI_DISABLE_GPU", "1") + clear_pyfock_gpu_probe_cache() + + assert probe_pyfock_gpu()[0] is False + assert "QUANTUI_DISABLE_GPU" in probe_pyfock_gpu()[2] + + +def test_per_calculation_opt_out_does_not_probe(monkeypatch): + def fail_probe(): + raise AssertionError("GPU probe should not run for explicit opt-out") + + monkeypatch.setattr("quantui.pyfock_gpu.probe_pyfock_gpu", fail_probe) + assert resolve_pyfock_gpu(False) == ( + False, + None, + "GPU acceleration disabled for this calculation", + )