diff --git a/INSTALL.md b/INSTALL.md new file mode 100644 index 0000000..e1ea188 --- /dev/null +++ b/INSTALL.md @@ -0,0 +1,205 @@ +# Installing sbd-eigensolver + +Every install compiles the extension modules on the target machine — no wheels are +published — so which backends you end up with depends on the toolchain the build +finds. The quickstart in the [README](README.md#installation) covers the common CPU +case; this page is the full matrix: prerequisites, the environment variables the build +reads, building against a host MPI, and how to verify what you got. + +## Prerequisites + +**Required:** Python 3.10+, MPI (OpenMPI/MPICH), BLAS (OpenBLAS/MKL), pybind11, mpi4py, numpy, compiler with OpenMP. + +**On macOS:** Apple clang ships without OpenMP, so add `llvm-openmp` to the conda +environment. Homebrew's `libomp` is used as a fallback if the env has none. Pin the +compiler with `CC`/`CXX` as well — a bare `clang++` is resolved through `PATH`, so a +Homebrew LLVM silently wins over both Apple clang and a conda toolchain. The build +prints which compiler and which libomp it chose. + +**For the GPU backends** — optional; without them you get a CPU-only install: + +***On NVIDIA*** (both the Thrust and the OpenMP-offload backend): + +- *to build:* [NVIDIA HPC SDK](https://developer.nvidia.com/hpc-sdk) (`nvc++`). + `SBD_GPU_ARCH` is **optional**: unset, nvc++ targets the GPU of the machine + the toolchain was installed on; set, it is honored exactly and may name + several generations at once (`cc80,cc90,cc100`) — see + [Environment Variables](#environment-variables). + +***On AMD*** (the OpenMP-offload backend only): + +- *to build:* ROCm LLVM toolchain (`amdclang++`). + `SBD_GPU_ARCH` is optional here on a GPU system and is **detected** with + ROCm's `amdgpu-arch` (e.g. `gfx90a` on MI250X, `gfx942` on MI300X). However, on a GPU-less build host, you must + set it. see [Environment Variables](#environment-variables). + +Either install path compiles the C++ extension on the target machine; +no pre-built wheels are published. The resulting binary depends on the +local MPI and BLAS, so both must be installed first. + +## Install from PyPI + +**A self-contained conda environment** is the quickest way to get those +dependencies in place for the CPU backend: +```bash +conda create -y -n sbd -c conda-forge \ + python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip mpi4py +# ...plus llvm-openmp on macOS +``` + +```bash +conda activate sbd +``` + +Now install the sbd-eigensolver-python package +``` +pip install sbd-eigensolver +``` + +The published source distribution (sdist) bundles the sbd header files, so this +needs no git checkout and no submodule step. + +## Install from git checkout + +Here the headers come from upstream +[r-ccs-cms/sbd](https://github.com/r-ccs-cms/sbd) via a git submodule at +`vendor/sbd-upstream/`, **pinned at a specific upstream commit**. Run `git submodule status` to see the pinned SHA. + +When installing from a git checkout, it is important to make sure the +sbd submodule is cloned, too: + +```bash +git clone --recurse-submodules https://github.com/Qiskit/sbd-eigensolver-python.git +# or, if you cloned without --recurse-submodules: +git submodule update --init --recursive +``` + +If you need a newer upstream revision (for a recently-landed GPU fix +etc.), advance the local submodule and rebuild: + +```bash +git submodule update --remote vendor/sbd-upstream +``` + +After the upstream sbd code is cloned, run: +``` +pip install -e . --no-build-isolation --force-reinstall --no-deps +``` + +## Environment Variables + +```bash +# --- NVIDIA GPU backends (Thrust and OpenMP target-offload): point at NVHPC. +# Only needed if nvc++ is not already on PATH. Adjust the path. +# Either the compilers directory or the version root above it works, and +# NVHPC_ROOT -- which NVIDIA's own modulefile sets to the version root -- is +# read too, so `module load nvhpc` alone is enough. +export NVHPC_HOME=/opt/nvidia/hpc_sdk/Linux_x86_64/2025/compilers + +# --- AMD GPU backend (OpenMP target-offload): point at ROCm LLVM toolchain +# amdclang++ is found on $ROCM_HOME/bin +export ROCM_HOME=/opt/rocm + +# --- OPTIONAL. Only for a host with BOTH GPU toolchains installed, where the +# auto-answer would be an accident of probe order: nvidia | amd | none +export SBD_GPU_VENDOR=amd + +# --- OPTIONAL GPU architecture, spelled per vendor. +# NVIDIA: unset, nvc++ targets the GPU of the machine this toolchain was +# installed on. Set it to pin the target, including SEVERAL at once: +# A100: cc80 H100: cc90 GB200 / B200: cc100 +# ccall-major one target per major generation +# AMD: unset, the arch is DETECTED with ROCm's `amdgpu-arch`. Set it to pin, +# or when building on a host with no GPU (where detection cannot work and +# the build stops asking for it): +# MI250X: gfx90a MI300X: gfx942 (several: gfx90a,gfx942) +export SBD_GPU_ARCH=cc80,cc90,cc100 # NVIDIA +export SBD_GPU_ARCH=gfx90a # AMD + +# --- optional overrides; each has a working default --- +# Which backends to build: defaults to CPU always, plus every GPU backend the +# detected toolchain supports -- Thrust AND OpenMP-offload under nvc++, +# OpenMP-offload only under amdclang++ (there is no rocThrust path). +# Set it only to narrow that: +# cpu CPU only -- skip GPU even if a GPU compiler is present +# gpu Thrust GPU only, no CPU -- NVIDIA only, errors on AMD +# gpu_omp_offload OpenMP target-offload GPU only (either vendor) +export SBD_BUILD_BACKEND=cpu + +# MPI: defaults to whatever mpi4py is linked against. Set this only +# for layouts that cannot be inferred. +# NOTE: Every GPU backend hands MPI device pointers, so use a GPU-aware MPI: +# CUDA-aware on NVIDIA, ROCm-aware on AMD. +export MPI_HOME=/path/to/mpi + +# BLAS: defaults to whatever the linker finds, including a +# conda-installed OpenBLAS in $CONDA_PREFIX/lib. Set these to select +# a specific build (e.g. an arch-tuned OpenBLAS) +export BLAS_LIB_PATH=/path/to/blas/lib +export BLAS_LIBS=openblas # or mkl_rt + +# these are read while COMPILING, so set them first, then install +pip install sbd-eigensolver # from PyPI +pip install -e . --no-build-isolation --no-deps # from a git checkout +``` + +## Build Using the Host MPI +```bash +# Create a conda env +conda create -y -n sbd -c conda-forge \ + python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip +# ...plus llvm-openmp on macOS + +conda activate sbd # always activate first + +# Install mpi4py against the host MPI +# For any GPU backend the host MPI must be GPU-aware: +# CUDA-aware on NVIDIA, ROCm-aware on AMD. +export MPI_HOME=/path/to/mpi +MPICC=$MPI_HOME/bin/mpicc python -m pip install --no-binary=mpi4py --no-cache-dir mpi4py +# NOTE: If no host MPI is available, let conda pick a compatible one with the +# command below. A default conda-forge MPI is not GPU-aware, which is fine for the +# CPU backend; for the GPU backends either install mpi4py against a GPU-aware MPI +# as above, or see the README's Backend Architecture section for the two build-time options that +# let the Thrust backend run without one. +# conda install -y -c conda-forge mpi4py + +# confirm which MPI mpi4py uses -- setup.py builds against exactly this +python -c "from mpi4py import MPI; print(MPI.Get_library_version())" + +# only for the SQD examples (examples/tpb/run_sqd_sbd.py and .ipynb) +pip install "qiskit-addon-sqd>=0.13.1" + +# install sbd-eigensolver +pip install sbd-eigensolver + +``` + +## Verify + +```bash +python -c "import sbd; print(sbd.available_backends())" +# CPU only: ['cpu'] +# NVIDIA, default build: ['cpu', 'gpu', 'gpu-omp'] +# AMD, default build: ['cpu', 'gpu-omp'] +# OMP-offload-only install: ['gpu-omp'] +``` + +On a GPU build, confirm which vendor and architecture the offload backend +targets — one `'gpu-omp'` device serves both vendors, so the name alone does not +say: + +```bash +python -c "import sbd; print(sbd.get_backend('gpu-omp').__sbd_offload_target__)" +# amdgcn-amd-amdhsa:gfx90a AMD MI250X +# nvptx64-nvidia-cuda:cc90 NVIDIA H100 +``` + +The Thrust backend is stamped too (`cuda:cc90`); the CPU backend reports `None`. + +## See Also + +- [README](README.md) — what the package is, the API, and the SQD integration +- [`examples/tpb/README.md`](examples/tpb/README.md) — backend selection at runtime, + bundled test data and performance notes +- [Troubleshooting](README.md#troubleshooting) — build and runtime symptoms diff --git a/MANIFEST.in b/MANIFEST.in index f8bddd5..ff153aa 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -16,4 +16,5 @@ recursive-include vendor/sbd-upstream/include *.h include vendor/sbd-upstream/LICENSE.txt # Example scripts and notebook, referenced by the README. -recursive-include python/examples *.py *.ipynb *.md +recursive-include examples *.py *.ipynb *.md *.json +include INSTALL.md diff --git a/README.md b/README.md index 5664190..e091a44 100644 --- a/README.md +++ b/README.md @@ -23,204 +23,41 @@ In addition to TPB, this package also contains experimental support for SBD's ** ## Installation -### Prerequisites +The extension modules are compiled on your machine — no wheels are published — so the +backends you get depend on the toolchain the build finds. For the common CPU case: -**Required:** Python 3.10+, MPI (OpenMPI/MPICH), BLAS (OpenBLAS/MKL), pybind11, mpi4py, numpy, compiler with OpenMP. - -**On macOS:** Apple clang ships without OpenMP, so add `llvm-openmp` to the conda -environment. Homebrew's `libomp` is used as a fallback if the env has none. Pin the -compiler with `CC`/`CXX` as well — a bare `clang++` is resolved through `PATH`, so a -Homebrew LLVM silently wins over both Apple clang and a conda toolchain. The build -prints which compiler and which libomp it chose. - -**For the GPU backends** — optional; without them you get a CPU-only install: - -***On NVIDIA*** (both the Thrust and the OpenMP-offload backend): - -- *to build:* [NVIDIA HPC SDK](https://developer.nvidia.com/hpc-sdk) (`nvc++`). - `SBD_GPU_ARCH` is **optional**: unset, nvc++ targets the GPU of the machine - the toolchain was installed on; set, it is honored exactly and may name - several generations at once (`cc80,cc90,cc100`) — see - [Environment Variables](#environment-variables). - -***On AMD*** (the OpenMP-offload backend only): - -- *to build:* ROCm LLVM toolchain (`amdclang++`). - `SBD_GPU_ARCH` is optional here on a GPU system and is **detected** with - ROCm's `amdgpu-arch` (e.g. `gfx90a` on MI250X, `gfx942` on MI300X). However, on a GPU-less build host, you must - set it. see [Environment Variables](#environment-variables). - -Either install path compiles the C++ extension on the target machine; -no pre-built wheels are published. The resulting binary depends on the -local MPI and BLAS, so both must be installed first. - -### Install from PyPI - -**A self-contained conda environment** is the quickest way to get those -dependencies in place for the CPU backend: ```bash conda create -y -n sbd -c conda-forge \ python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip mpi4py # ...plus llvm-openmp on macOS -``` - -```bash conda activate sbd -``` -Now install the sbd-eigensolver-python package -``` pip install sbd-eigensolver ``` -The published source distribution (sdist) bundles the sbd header files, so this -needs no git checkout and no submodule step. - -### Install from git checkout - -Here the headers come from upstream -[r-ccs-cms/sbd](https://github.com/r-ccs-cms/sbd) via a git submodule at -`vendor/sbd-upstream/`, **pinned at a specific upstream commit**. Run `git submodule status` to see the pinned SHA. - -When installing from a git checkout, it is important to make sure the -sbd submodule is cloned, too: - -```bash -git clone --recurse-submodules https://github.com/Qiskit/sbd-eigensolver-python.git -# or, if you cloned without --recurse-submodules: -git submodule update --init --recursive -``` - -If you need a newer upstream revision (for a recently-landed GPU fix -etc.), advance the local submodule and rebuild: - -```bash -git submodule update --remote vendor/sbd-upstream -``` - -After the upstream sbd code is cloned, run: -``` -pip install -e . --no-build-isolation --force-reinstall --no-deps -``` - -### Environment Variables - -```bash -# --- NVIDIA GPU backends (Thrust and OpenMP target-offload): point at NVHPC. -# Only needed if nvc++ is not already on PATH. Adjust the path. -export NVHPC_HOME=/opt/nvidia/hpc_sdk/Linux_x86_64/2025/compilers - -# --- AMD GPU backend (OpenMP target-offload): point at ROCm LLVM toolchain -# amdclang++ is found on $ROCM_HOME/bin -export ROCM_HOME=/opt/rocm - -# --- OPTIONAL. Only for a host with BOTH GPU toolchains installed, where the -# auto-answer would be an accident of probe order: nvidia | amd | none -export SBD_GPU_VENDOR=amd - -# --- OPTIONAL GPU architecture, spelled per vendor. -# NVIDIA: unset, nvc++ targets the GPU of the machine this toolchain was -# installed on. Set it to pin the target, including SEVERAL at once: -# A100: cc80 H100: cc90 GB200 / B200: cc100 -# ccall-major one target per major generation -# AMD: unset, the arch is DETECTED with ROCm's `amdgpu-arch`. Set it to pin, -# or when building on a host with no GPU (where detection cannot work and -# the build stops asking for it): -# MI250X: gfx90a MI300X: gfx942 (several: gfx90a,gfx942) -export SBD_GPU_ARCH=cc80,cc90,cc100 # NVIDIA -export SBD_GPU_ARCH=gfx90a # AMD - -# --- optional overrides; each has a working default --- -# Which backends to build: defaults to CPU always, plus every GPU backend the -# detected toolchain supports -- Thrust AND OpenMP-offload under nvc++, -# OpenMP-offload only under amdclang++ (there is no rocThrust path). -# Set it only to narrow that: -# cpu CPU only -- skip GPU even if a GPU compiler is present -# gpu Thrust GPU only, no CPU -- NVIDIA only, errors on AMD -# gpu_omp_offload OpenMP target-offload GPU only (either vendor) -export SBD_BUILD_BACKEND=cpu - -# MPI: defaults to whatever mpi4py is linked against. Set this only -# for layouts that cannot be inferred. -# NOTE: Every GPU backend hands MPI device pointers, so use a GPU-aware MPI: -# CUDA-aware on NVIDIA, ROCm-aware on AMD. -export MPI_HOME=/path/to/mpi - -# BLAS: defaults to whatever the linker finds, including a -# conda-installed OpenBLAS in $CONDA_PREFIX/lib. Set these to select -# a specific build (e.g. an arch-tuned OpenBLAS) -export BLAS_LIB_PATH=/path/to/blas/lib -export BLAS_LIBS=openblas # or mkl_rt - -# these are read while COMPILING, so set them first, then install -pip install sbd-eigensolver # from PyPI -pip install -e . --no-build-isolation --no-deps # from a git checkout -``` - -### Build Using the Host MPI -```bash -# Create a conda env -conda create -y -n sbd -c conda-forge \ - python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip -# ...plus llvm-openmp on macOS - -conda activate sbd # always activate first - -# Install mpi4py against the host MPI -# For any GPU backend the host MPI must be GPU-aware: -# CUDA-aware on NVIDIA, ROCm-aware on AMD. -export MPI_HOME=/path/to/mpi -MPICC=$MPI_HOME/bin/mpicc python -m pip install --no-binary=mpi4py --no-cache-dir mpi4py -# NOTE: If no host MPI is available, let conda pick a compatible one with the -# command below. A default conda-forge MPI is not GPU-aware, which is fine for the -# CPU backend; for the GPU backends either install mpi4py against a GPU-aware MPI -# as above, or see "Backend Architecture" below for the two build-time options that -# let the Thrust backend run without one. -# conda install -y -c conda-forge mpi4py - -# confirm which MPI mpi4py uses -- setup.py builds against exactly this -python -c "from mpi4py import MPI; print(MPI.Get_library_version())" - -# only for the SQD examples (python/examples/run_sqd_sbd.py and .ipynb) -pip install "qiskit-addon-sqd>=0.13.1" - -# install sbd-eigensolver -pip install sbd-eigensolver - -``` - -### Verify +Then check what was built: ```bash python -c "import sbd; print(sbd.available_backends())" -# CPU only: ['cpu'] -# NVIDIA, default build: ['cpu', 'gpu', 'gpu-omp'] -# AMD, default build: ['cpu', 'gpu-omp'] -# OMP-offload-only install: ['gpu-omp'] +# CPU only: ['cpu'] +# NVIDIA, default build: ['cpu', 'gpu', 'gpu-omp'] +# AMD, default build: ['cpu', 'gpu-omp'] ``` -On a GPU build, confirm which vendor and architecture the offload backend -targets — one `'gpu-omp'` device serves both vendors, so the name alone does not -say: - -```bash -python -c "import sbd; print(sbd.get_backend('gpu-omp').__sbd_offload_target__)" -# amdgcn-amd-amdhsa:gfx90a AMD MI250X -# nvptx64-nvidia-cuda:cc90 NVIDIA H100 -``` - -The Thrust backend is stamped too (`cuda:cc90`); the CPU backend reports `None`. +**See [INSTALL.md](INSTALL.md)** for the rest: prerequisites per platform, installing +from a git checkout with the submodule, every environment variable the build reads +(GPU toolchains, architectures, MPI and BLAS selection, narrowing which backends get +built), building against an existing host MPI, and fuller verification. ## Examples -Located in `python/examples/`: - -- [`run_sbd_diag.py`](python/examples/run_sbd_diag.py) — Standalone TPB diagonalization (no Qiskit dependency) -- [`run_sqd_sbd.ipynb`](python/examples/run_sqd_sbd.ipynb) — Jupyter Notebook SQD loop with SBD solver (random or hardware bitstrings) -- [`run_sqd_sbd.py`](python/examples/run_sqd_sbd.py) — SQD loop with SBD solver (random or hardware bitstrings) -- [`run_sqd_enlarge_subspace_sbd.py`](python/examples/run_sqd_enlarge_subspace_sbd.py) — SQD that also grows its own subspace between rounds via single excitations +Located in `examples/`, organized by basis type since the solvers take different +subspaces and decompose over MPI differently. Each folder's README is the authoritative +guide to what it contains, how to run it, and the backend and threading settings that +matter for it. -See [python/examples/README.md](python/examples/README.md) for usage details. +- [`examples/tpb/`](examples/tpb/README.md) — tensor-product basis: standalone TPB + diagonalization, the SQD loops, and the subspace-enlargement driver. ## Integration with qiskit-addon-sqd @@ -255,9 +92,9 @@ result = diagonalize_fermionic_hamiltonian( ) ``` -See [SQD Parameters](python/examples/README.md#sqd-parameters) for how each +See [SQD Parameters](examples/tpb/README.md#sqd-parameters) for how each parameter feeds the loop, and -[python/examples/run_sqd_sbd.py](python/examples/run_sqd_sbd.py) for a +[examples/tpb/run_sqd_sbd.py](examples/tpb/run_sqd_sbd.py) for a complete example. qiskit-addon-sqd is the orchestrator in that recipe: it owns the loop @@ -267,7 +104,7 @@ say in how the subspace grows between iterations. ### SQD with subspace enlargement -[`run_sqd_enlarge_subspace_sbd.py`](python/examples/run_sqd_enlarge_subspace_sbd.py) +[`run_sqd_enlarge_subspace_sbd.py`](examples/tpb/run_sqd_enlarge_subspace_sbd.py) builds on the same recipe, but grows its own subspace between rounds: after each solve, it expands the dominant determinant pairs via qiskit-addon-sqd's own `enlarge_batch_from_transitions` (same-spin single excitations, both @@ -277,7 +114,7 @@ alpha and beta) and feeds the result forward as the next round's own outer Python loop, rather than delegating the whole multi-iteration loop to one call — that is what makes injecting a step between rounds possible. -On the bundled H2O pool ([`count_dict_h2o.json`](python/examples/count_dict_h2o.json), +On the bundled H2O pool ([`count_dict_h2o.json`](examples/tpb/count_dict_h2o.json), 275 bitstrings), plain SQD reaches ≈ -76.236 Ha and stops there; this driver keeps going past that fixed pool on its own and converges to **-76.2421767512 Ha**. @@ -361,13 +198,32 @@ and replaces the determinant communicators with a single basis communicator: | Attribute | Default | Description | |-----------|---------|-------------| -| `b_comm_size` | 1 | Basis communicator size (must be 1 for `gdb_diag`) | -| `t_comm_size` | 1 | Task communicator size | +| `b_comm_size` | 1 | Basis communicator size — must be 1 for `gdb_diag`, see below | +| `t_comm_size` | 1 | Task communicator size — must be 1 while `b_comm_size` is, see below | | `seed` | 1729 | Seed for the initial vector | | `heatbath_cutoff` | 1e-4 | Heatbath expansion cutoff | | `heatbath_truncation` | 0.0 | Weight truncation applied before heatbath expansion | | `heatbath_batch_size` | 200000000 | Heatbath expansion batch size | +**Why both must be 1**, since the two constraints have different owners: + +`b_comm_size == 1` is a limitation of *this wrapper*, not of SBD. Upstream's in-memory +`gdb::diag` expects each rank to pass **its own shard** of the determinant list — that +is what upstream's file-based entry point hands it, after distributing determinant +files across `b_comm`. This wrapper passes the whole list from every rank, which is +only consistent with a single basis block, so it rejects anything else rather than +have each rank diagonalize the full subspace while believing it held a shard. Upstream +itself runs with a split basis: its own `run.sh` for the GDB app passes +`--b_comm_size 2`. + +`t_comm_size == 1` then follows from *upstream's* algorithm rather than from us. GDB's +matrix-vector product rotates the ket around `b_comm` as a ring, so there are exactly +`b_comm_size` ring stations and one "task" is one station — meaning `t_comm_size` +cannot exceed `b_comm_size`. With the basis in a single block there is a single task. + +Ranks are not wasted in the meantime: the derived helper dimension, +`ranks / (t_comm_size × b_comm_size)`, absorbs them and does not change the energy. + ### Diagonalization ```python @@ -409,16 +265,42 @@ The optional `device` parameter overrides the default set by `init()`. the build respects a caller-set compiler, so `nvc++`/`amdclang++` never run. Unset `CC`/`CXX`, or keep conda compilers out of the build env. -**GPU not building:** On NVIDIA check `which nvc++` and set `NVHPC_HOME`. On AMD +**GPU not building:** On NVIDIA check `which nvc++` and set `NVHPC_HOME` (or rely on +`NVHPC_ROOT` from `module load nvhpc`; either the compilers directory or the version +root works). On AMD check `which amdclang++` and set `ROCM_HOME`. The build prints which toolchain it picked (`Found amdclang++ in PATH: …` / `Found NVIDIA HPC SDK at: …`) and, for the offload backend, the resolved architecture; on a host with both toolchains force the choice with `SBD_GPU_VENDOR=amd|nvidia`. -**MPI errors:** Verify `MPI_HOME`, check `python -c "from mpi4py import MPI; print(MPI.Get_version())"`. - -**OMP-offload runs all land on GPU 0 in multi-GPU jobs:** symptom — every MPI rank shows large memory only on GPU 0 in `nvidia-smi` (or `rocm-smi`). The bindings call `omp_set_default_device(mpi_rank % n_dev)`, but `omp_get_num_devices()` can return 0 in some dlopen scenarios. The bindings fall back to counting the entries in the vendor's device-visibility variable — `CUDA_VISIBLE_DEVICES` on NVIDIA, `ROCR_VISIBLE_DEVICES` or `HIP_VISIBLE_DEVICES` on AMD — so make sure the relevant one is exported and lists all your GPUs (e.g. `0,1,2,3`). Slurm/`srun --gres=gpu:N` and OpenMPI's default binding policy already do this; if you've custom-restricted it to a single GPU per rank, set it manually before launch. +**`MPI_HOME=... is not the MPI that mpi4py is linked against`:** the build stops here +deliberately rather than producing extensions that link one MPI while `mpi4py` loads +another — a mismatch that surfaces later as undefined symbols or a hang inside the first +collective. `MPI_HOME` is only for layouts the build cannot infer; unset it to use +`mpi4py`'s own MPI, or reinstall `mpi4py` against the MPI you want +(`pip install --no-binary :all: mpi4py`). To see which MPI that is: +`python -c "from mpi4py import MPI; print(MPI.Get_library_version())"`. + +**`OMP: Error #15: Initializing libomp.dylib, but found libomp.dylib already +initialized` on macOS:** two copies of the same LLVM OpenMP runtime in one process. It +aborts at the first parallel region, so the import succeeds and the first +diagonalization dies. Usually it means the environment provides `libomp` twice — for +example Homebrew `llvm` *and* Homebrew `libomp`, or a Homebrew copy alongside the conda +env's. Build against one only; the conda env's is the one loaded at import time, so +prefer it. Tracked as +[issue #27](https://github.com/Qiskit/sbd-eigensolver-python/issues/27). + +**OMP-offload runs all land on GPU 0 in multi-GPU jobs:** symptom — every MPI rank shows large memory only on GPU 0 in `nvidia-smi` (or `rocm-smi`). The bindings call `omp_set_default_device(mpi_rank % n_dev)`, but `omp_get_num_devices()` can return 0 in some dlopen scenarios. The bindings fall back to counting the entries in the vendor's device-visibility variable — `CUDA_VISIBLE_DEVICES` on NVIDIA, `ROCR_VISIBLE_DEVICES` or `HIP_VISIBLE_DEVICES` on AMD — so make sure the relevant one is exported and lists all your GPUs (e.g. `0,1,2,3`). Slurm/`srun --gres=gpu:N` and OpenMPI's default binding policy already do this; if you've custom-restricted it to a single GPU per rank, set it manually before launch. Note the index is the **global** MPI rank, not a node-local one, so the assignment is even only when the launcher places ranks on nodes in contiguous blocks — round-robin placement leaves each node using a strided subset of its GPUs. **Ranks die with `Bus error` or `SIGSEGV` inside the MPI's own copy path** (`MPIR_Localcopy`, `ucp_worker_progress`, ...) **on a GPU backend:** the MPI is not GPU-aware and was handed a device pointer. Rebuild UCX `--with-cuda` / `--with-rocm`, and confirm with `ucx_info -d | grep -i 'Transport: cuda'` (or `rocm`). Two things mislead here. A partly GPU-aware stack fails in only one place: an MPICH with GPU support *disabled* over a CUDA-aware UCX ran OMP-offload fine and crashed only in Thrust, because the inter-rank path went through UCX while the local-copy path did not. And on AMD a non-ROCm-aware MPI does not crash at all — ROCm maps device memory into the process address space, so the host copy succeeds and merely stages everything through the host aperture (measured on MI250X, XNACK off, 8 ranks) — so a working AMD run is not evidence that the MPI is ROCm-aware. +**`GDB Thrust mult does not support h_comm_size > 1` from `gdb_diag` on more than one +rank:** GDB's Thrust kernels never implemented the helper dimension, and the helper +dimension is `ranks / (t_comm_size × b_comm_size)`. Since `gdb_diag` requires +`b_comm_size == 1` (see [Configuration](#configuration)), which forces `t_comm_size` to +1, every rank you add lands in the helper dimension — so GPU GDB is limited to a single +rank in this release. Run GDB on one GPU, or on the CPU backend, where the helper +dimension is unconstrained. TPB is unaffected and shards over `adet_comm_size` / +`bdet_comm_size` as usual. + **Repository:** https://github.com/Qiskit/sbd-eigensolver-python diff --git a/examples/tpb/README.md b/examples/tpb/README.md new file mode 100644 index 0000000..3efe7d3 --- /dev/null +++ b/examples/tpb/README.md @@ -0,0 +1,411 @@ +# TPB examples — SQD/SBD with a tensor-product basis + +Examples demonstrating SBD's capabilities for quantum chemistry calculations. + +## Overview + +These examples all use TPB, where the subspace is the Cartesian product of an alpha and +a beta determinant list. + +## Examples + +### 1. run_sbd_diag.py — Standalone SBD Diagonalization + +Runs a single TPB diagonalization from an FCIDUMP file and alpha determinant +file. No SQD loop, no Qiskit dependency. + +```bash +# H2O with 2 MPI ranks with CPU +mpirun -np 2 python -u run_sbd_diag.py \ + --device cpu \ + --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ + --adetfile ../../vendor/sbd-upstream/data/h2o/h2o-1em3-alpha.txt \ + --adet_comm_size 2 + +# N2 with GPU +mpirun -np 8 python -u run_sbd_diag.py \ + --device gpu \ + --fcidump ../../vendor/sbd-upstream/data/n2/fcidump.txt \ + --adetfile ../../vendor/sbd-upstream/data/n2/1em3-alpha.txt \ + --adet_comm_size 4 --bdet_comm_size 2 + +# Retrieve the 1-/2-particle RDMs and save them to a file +mpirun -np 2 python -u run_sbd_diag.py --rdm_output /tmp/h2o_rdms.npz +``` + +`--rdm_output` takes the file to save to (`/tmp/h2o_rdms.npz` above) and +writes **one** `.npz` file there holding both `rdm1` and `rdm2` together +(`data = np.load("/tmp/h2o_rdms.npz"); data["rdm1"]`, `data["rdm2"]`) — +unlike upstream SBD's own CLI, which writes two separate files +(`1pRDM.txt`/`2pRDM.txt`). It also prints `trace(rdm1)` and the natural +orbital occupations. Leaving it empty (the default) skips computing RDMs +entirely. + +By default beta determinants are derived from `--adetfile` alone (identical +to it, or a shuffled copy if `--shuffle` is set). `--symmetrize_spin 0` +loads `--adetfile` and `--bdetfile` as independent, genuinely distinct +alpha/beta determinant sets instead; `--bdetfile` is otherwise ignored +(with a warning) since symmetric mode always derives beta from alpha. + +**Key options:** `--device`, `--fcidump`, `--adetfile`, `--bdetfile`, +`--symmetrize_spin`, `--adet_comm_size`, `--bdet_comm_size`, +`--task_comm_size`, `--method`, `--tolerance`, `--iteration`, +`--rdm_output`. (These keep their unprefixed names here: this driver *is* +SBD. The SQD drivers prefix them `--sbd_*`.) Run `python run_sbd_diag.py +--help` for the full list. + +**Requirements:** `sbd`, `mpi4py` + +### 2. run_sqd_sbd.py — SQD Loop with SBD Solver + +Runs the self-consistent SQD workflow (qiskit-addon-sqd) using SBD as the +eigensolver backend. Supports two bitstring input modes: + +- `--counts FILE` — load bitstrings from a count_dict.json +- `--samples N` — generate N random bitstrings at the target Hamming weights + (default). A plumbing check only: random determinants give a random subspace, + so the energy is not meaningful. Use `--counts` for real results. + +```bash +# H2O with the bundled counts file (275 bitstrings -> ~ -76.236 Ha) +mpirun -np 8 python -u run_sqd_sbd.py \ + --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ + --counts count_dict_h2o.json \ + --device gpu \ + --adet_comm_size 4 --bdet_comm_size 2 + +# Custom system with random bitstrings +mpirun -np 8 python -u run_sqd_sbd.py \ + --fcidump /path/to/fci_dump.txt \ + --samples_per_batch 800 --num_batches 3 --max_iterations 10 \ + --device gpu \ + --adet_comm_size 4 --bdet_comm_size 2 +``` + +**count_dict.json format:** A JSON object mapping bitstrings to shot counts, as +produced by a quantum device or simulator. Each bitstring has length `2 × NORB` +and is laid out as **`[beta | alpha]`**: the first `NORB` bits are beta +(spin-down), the last `NORB` are alpha (spin-up), and within each half **orbital 0 +is the rightmost bit**. qiskit-addon-sqd postselects the last `NORB` bits on +`num_elec_a` and the first `NORB` on `num_elec_b`. + +For H2O (NORB=24, 5α+5β) the Hartree–Fock configuration — the five lowest +orbitals doubly occupied — is therefore `"0"*19 + "1"*5` in *both* halves: + +```json +{ + "000000000000000000011111000000000000000000011111": 16, + "010000000010001010000001010000000001000010100100": 12, + "000010001110000000000010001001000110000000000100": 8 +} +``` + +Bitstrings whose halves do not hold exactly `num_elec_a` / `num_elec_b` ones are +dropped by postselection, so a file of uniform-random strings yields nothing +usable — for H2O only `C(24,5)² / 4²⁴ ≈ 6e-6` of them qualify. + +[`count_dict_h2o.json`](./count_dict_h2o.json) in this directory is a ready-made +H2O example: 275 bitstrings taken from the vendored `h2o-1em3-alpha.txt` +determinant list, giving a 275 × 275 = 75,625-determinant subspace at +≈ -76.236 Ha. + +**Key options:** `--fcidump` (required), `--counts`, `--samples`, +`--samples_per_batch`, `--num_batches`, `--max_iterations`, `--device`, +MPI decomposition flags, and the SQD tolerances `--energy_tol` / +`--occupancies_tol` / `--sqd_carryover_threshold`. Inner-solver flags are prefixed +(`--sbd_eps`, `--sbd_max_it`, `--sbd_method`, ...) and have sensible defaults; the +old unprefixed spellings still work. Run `python run_sqd_sbd.py --help` for the +full list, which is grouped by layer. + +**Requirements:** see [Integration with qiskit-addon-sqd](../../README.md#integration-with-qiskit-addon-sqd) in the Python Bindings README (`pyscf`, `qiskit`, `qiskit-addon-sqd`). + +See [SQD Parameters](#sqd-parameters) below for the full reference, grouped by SQD loop / SBD solver / MPI grid / checkpointing. + +### 3. run_sqd_enlarge_subspace_sbd.py — SQD that grows its own subspace + +Same self-consistent SQD loop as `run_sqd_sbd.py` above (sampling, +configuration recovery, SBD as the solver), but with one addition: between +rounds, it expands the dominant determinant pairs from the just-solved +wavefunction via qiskit-addon-sqd's own `enlarge_batch_from_transitions` +(same-spin single-electron excitations, both alpha and beta), and feeds the +result forward as the next round's `include_configurations`. Concretely, +it calls `diagonalize_fermionic_hamiltonian` with `max_iterations=1` itself, +in its own outer Python loop, rather than delegating the whole +multi-iteration loop to one call the way `run_sqd_sbd.py` does -- that's +what makes injecting a step between rounds possible. The loop stops when +either the expanded set adds nothing new, or the energy and occupancies +both stop moving (`--energy_tol`/`--occupancies_tol`) -- `--max_iterations` +is a safety cap, not the expected stopping mechanism. + +```bash +mpirun -np 8 python -u run_sqd_enlarge_subspace_sbd.py \ + --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ + --counts count_dict_h2o.json \ + --device gpu \ + --adet_comm_size 4 --bdet_comm_size 2 --enlarge_threshold 1e-4 +``` + +**A note on JAX and GPUs.** The expansion step above is JAX. Since XLA reserves 75% of a device +on first use, one rank claims it and the rest fail with `RESOURCE_EXHAUSTED: CUDA_ERROR_OUT_OF_MEMORY`. +On a GPU backend this driver therefore does two things for you, so the command above needs no extra environment: + +- assigns each rank its own device, using `sbd.get_device_id()` so JAX lands + on the same card SBD already selected for that rank; +- sets `XLA_PYTHON_CLIENT_PREALLOCATE=false`, so XLA takes memory on demand + instead of reserving 75% of the card away from SBD. Without this, SBD's + Davidson basis can run out of room at large `--max_dim` and fail inside + `tpb_diag` with `std::bad_alloc` -- a memory error that reads as SBD's + fault but is caused by JAX's reservation. + +`XLA_PYTHON_CLIENT_PREALLOCATE` is left alone if you set it yourself, and +`JAX_PLATFORMS=cpu` still forces the expansion onto CPU. The device assignment +always runs, and stays correct if you pin one GPU per rank yourself: with a +single card visible the count is 1, so every rank resolves to device 0 -- its +own. + +Same bundled 275-bitstring H2O pool as `run_sqd_sbd.py`'s own example above: +plain SQD reaches **≈ -76.236 Ha** and stops there; this driver keeps going +past that fixed pool on its own and converges to **-76.2421767512 Ha**. + +See [SQD Parameters](#sqd-parameters) below for the flags it shares with +`run_sqd_sbd.py` and the ones that differ (`--enlarge_threshold` in place +of `--sqd_carryover_threshold`, and `--max_dim`'s risk profile is sharper +here). + +### 4. run_sqd_sbd.ipynb — Jupyter walkthrough (serial) + +Interactive single-rank companion to `run_sqd_sbd.py`. Same SQD self-consistent +loop on h2o, but inside a Jupyter kernel (`MPI.COMM_WORLD` size 1). Uses the +bundled [`count_dict_h2o.json`](./count_dict_h2o.json) (275 bitstrings → 75,625 +determinants) and reaches ≈ −76.236 Ha in a few seconds on CPU. + +```bash +pytest --nbmake run_sqd_sbd.ipynb # what CI runs; needs the nbtest extra +# or open it in JupyterLab and step through the cells +``` + +## SQD Parameters + +Reference for every flag `run_sqd_sbd.py` accepts, grouped the way `--help` +groups them: SQD loop, SBD solver, MPI grid, checkpointing. + +**How each iteration builds its subspace.** SQD samples bitstrings from a +quantum device, repairs the noisy ones against an orbital-occupancy estimate +(**configuration recovery**), subsamples them into batches, and diagonalizes +each batch. What makes it a *loop* is that two results feed back into the next +iteration. Three sources, concatenated in this priority order inside +qiskit-addon-sqd's `diagonalize_fermionic_hamiltonian` (`fermion.py`): + +``` +strs_a = include_a ++ carryover_strings_a ++ samples_a then dedupe, truncate to max_dim, sort +``` + +1. **`include_a`** — configurations passed as `include_configurations`. Static: + fixed before the loop, present every iteration, never updated. +2. **`carryover_strings_a`** — from the *previous* iteration's wavefunction. Every + determinant whose `|coefficient|` is at least `--sqd_carryover_threshold` + survives, ranked by `|c|^2`. +3. **`samples_a`** — drawn fresh this iteration, sorted by marginal probability. + +The order matters when `max_dim` is set: `include` and `carryover` are kept ahead of +fresh samples, so if those two already fill the cap, this iteration's new samples are +truncated away entirely. + +The samples are not re-used raw counts. Each iteration re-runs configuration +recovery from the *original* bitstrings using the occupancies from the previous +iteration's best batch (`_prepare_ci_strings` in `fermion.py`), then +subsamples. Recovery is **not +cumulative** — it always re-derives from the raw samples, just with a better +occupancy estimate each time. On iteration 1 there are no occupancies yet, so the +raw samples are only filtered by electron count (Hamming-weight postselection). + +So exactly two things flow from iteration N to N+1, and neither is a tolerance: +the **average orbital occupancies** (into recovery, source 3) and the +**wavefunction amplitudes** (into carryover, source 2). + +### SQD loop parameters + +Shared by both `run_sqd_sbd.py` and `run_sqd_enlarge_subspace_sbd.py` except +where noted. **`--max_dim` is the one that most needs attention**: it has no +universally safe default (see below), and in `run_sqd_enlarge_subspace_sbd.py` +leaving it unset is riskier still, since each round's subspace can grow from +the previous one rather than being resampled at a fixed size — the driver +prints an OOM warning when it detects this. + +*Shapes the subspace — changes the numbers you compute:* + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--counts FILE` | Load hardware bitstrings from a JSON file (use this or `--samples`) | none — falls back to `--samples` if omitted | +| `--samples N` | Generate N random bitstrings at the target Hamming weights; plumbing check only, energy not meaningful | `3000` (only used when `--counts` is omitted) | +| `--samples_per_batch` | Dominant control on subspace dimension. With `--symmetrize_spin 1` the alpha and beta string sets are merged, so the subspace is up to `(2N)^2`, not `N^2` | `3000` | +| `--symmetrize_spin` | `1` (default): merge the alpha and beta string pools every iteration, forcing `ci_strs_a == ci_strs_b`. SBD itself supports distinct alpha/beta determinant sets — this is purely a qiskit-addon-sqd loop-layer setting. `0`: sample and carry over alpha and beta independently, allowing them to differ | `1` | +| `--num_batches` | Independent subsamples per iteration; occupancies are averaged across them | `1` (`run_sqd_enlarge_subspace_sbd.py`) / `3` (`run_sqd_sbd.py`) | +| `--sqd_carryover_threshold` | `run_sqd_sbd.py` only. `\|coefficient\|` cutoff for carrying a determinant into the next iteration's sample pool. **Lower it to carry more** | `1e-4` | +| `--enlarge_threshold` | `run_sqd_enlarge_subspace_sbd.py` only — the analogous "carry more" knob for that driver, but structurally different: it gates which *pairs* get expanded into single excitations via `enlarge_batch_from_transitions`, not which determinants survive into resampling. **Lower it to expand more pairs per round** | `1e-4` | +| `--max_dim` | **Critical.** Cap on strings per spin sector, so the subspace cannot exceed `max_dim^2`. The main brake on runaway cost — no fixed value is safe for every system, since the right cap depends on available memory and orbital count. Start from a value known to work at a similar orbital count (e.g. `15000` was used for a 45-orbital system) and adjust down if you see an OOM | unset (no cap) | +| `--include_hf` | Force the single Slater determinant with the lowest `num_elec_a`/`num_elec_b` orbital indices occupied into `include_configurations`, every iteration. Cheap correctness check: that determinant's own diagonal energy is an exact lower bound on what a subspace containing it can do — if forcing it in moves the result, the sampled pool was missing it (and probably its low-excitation neighbors too) | off | + +*Decides when to stop — changes nothing about the subspace:* + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--max_iterations` | Hard cap on loop iterations (not the inner `--sbd_max_it`). In `run_sqd_enlarge_subspace_sbd.py` this is a safety cap only — the loop normally stops earlier, once a round adds no new determinants or both tolerances below are met | `30` (`run_sqd_enlarge_subspace_sbd.py`) / `5` (`run_sqd_sbd.py`) | +| `--energy_tol` | Iteration-to-iteration change in energy | `1e-8` | +| `--occupancies_tol` | Largest change in any single orbital occupancy — an infinity norm, not an average | `1e-5` | + +**Both stopping criteria must hold in the same iteration.** `fermion.py`'s +convergence check combines the energy-change test and the occupancy-change test +with a logical `and`, so the loop only stops once both are satisfied at once — +not whichever one happens first. A run that reaches +`--max_iterations` may be converged in energy while one stubborn orbital's +occupancy is still moving, and loosening only one tolerance will not stop it. +Watch the per-batch energies: while they still disagree, the loop has not +converged regardless of what the total says. + +### SBD solver parameters + +Per diagonalization, not per loop: + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--sbd_eps` | Davidson stop: **norm of the residual vector**, not an energy. Error in the energy goes roughly as `\|R\|^2/gap`, so this already implies far better energy accuracy than `--energy_tol` asks for. Tighten it for a near-degenerate system | `1e-5` | +| `--sbd_max_it` | Cap on Davidson iterations. Reaching it before `--sbd_eps` returns a partially converged vector **with no warning** — watch the `tol=` values SBD prints, and cross-batch agreement | `10` | +| `--sbd_max_nb` | Davidson basis vectors (block size). Peak memory during Davidson grows with how many sub-iterations it actually needs, not just `--max_dim` — a run that succeeds for several iterations at a fixed `dim` can still later need more basis vectors and run out of memory even though the subspace itself did not grow. Lowering this trades some convergence robustness for a lower memory ceiling | `10` | +| `--sbd_method` | 0=Davidson, 1=Davidson+Ham, 2=Lanczos, 3=Lanczos+Ham | `0` | +| `--sbd_use_precalculated_dets` | Thrust only. `1` precomputes a determinant index for **every** (α,β) pair — the whole subspace, on the GPU. `0` uses per-thread storage: slower per matvec, far less memory | `1` | +| `--sbd_max_memory_gb_for_determinants` | Thrust only, and **only consulted when `--sbd_use_precalculated_dets 0`** (`mult_thrust.h:257-273`). Caps the per-thread buffer in GB | `-1` (uncapped) | + +For reference, upstream's own `TPB_SBD` struct defaults are looser still (`max_it=1`, +`eps=1e-4`), and `run_sbd_diag.py` uses `eps=1e-3`. On the h2o counts case, +`eps=1e-5` and `eps=1e-8` give the same energy to ten decimal places. + +### MPI grid parameters + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--adet_comm_size` | Ranks spanning the alpha-determinant dimension | `1` | +| `--bdet_comm_size` | Ranks spanning the beta-determinant dimension | `1` | +| `--task_comm_size` | Ranks spanning task-level parallelism | `1` | + +All ranks diagonalize each batch together, then move to the next batch +sequentially. See [MPI Decomposition](#mpi-decomposition) below for the full 4D +grid (a fourth, *derived* dimension — `helper` — is not set directly). + +### Checkpointing parameters + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--checkpoint_path` | Write `ci_strs_a`/`ci_strs_b`/`orbital_occupancies`/energy to this path as JSON text (rank 0 only), every `--checkpoint_frequency` iterations. **Must be visible under the same path from every rank** — `--resume_from` has no rank-0-reads-then-broadcasts step, every rank opens this path itself | unset | +| `--checkpoint_frequency` | Write every this many iterations, plus always on the last one regardless of alignment | `1` (every iteration) | +| `--resume_from` | Seed a new run's `include_configurations`/`initial_occupancies` from a previous `--checkpoint_path`'s last recorded iteration | unset | + +What's preserved is which determinants (`ci_strs_a`/`ci_strs_b`, as plain +integers — independent of MPI decomposition, since they carry no rank/grid +information) and the derived per-orbital occupancy averages. **Not** the +wavefunction amplitudes matrix itself (`dim_a × dim_b` floats — at large +`--max_dim` this alone would dwarf everything else; neither consuming parameter +needs it). + +Not a bit-identical continuation: RNG state is fresh in the new process, and +every string from the resumed iteration becomes a **permanent** include for +every iteration of the new run — unlike a true single-process continuation, +where `--sqd_carryover_threshold` would keep pruning low-weight determinants +each iteration. A resumed run is seeded richer than an actual continuation +would have been at that point, not identical to one. + +## MPI Decomposition + +Total MPI ranks must be a **multiple** of +`task_comm_size × adet_comm_size × bdet_comm_size` — not equal to it. SBD splits +the ranks you asked for across those three dimensions and puts whatever remains +into a fourth, "helper" dimension, computed as +`ranks / (task_comm_size × adet_comm_size × bdet_comm_size)`. + +So 8 ranks with `--adet_comm_size 2 --bdet_comm_size 2` is valid: the grid is +`1 × 2 × 2` and the helper dimension absorbs the remaining factor of 2. + +When using more than one rank, specify at least `--adet_comm_size`. Examples: + +| Ranks | Decomposition | Helper | +|-------|---------------|--------| +| 1 | default (all = 1) | 1 | +| 2 | `--adet_comm_size 2` | 1 | +| 4 | `--adet_comm_size 2 --bdet_comm_size 2` | 1 | +| 8 | `--adet_comm_size 2 --bdet_comm_size 2` | 2 | +| 8 | `--adet_comm_size 2 --bdet_comm_size 2 --task_comm_size 2` | 1 | + +## Expected Results + +- **H2O**: ground state energy ≈ **-76.236 Hartree** +- **N2**: ground state energy ≈ **-109.042 Hartree** (with 1e-3 dets) + +## Backend Selection + +Every backend the toolchain supported was compiled into this one install, and +each is imported **lazily, on first use** — normally one per process. Select +per-call via `--device`: + +```bash +--device cpu # host OpenMP (default) +--device gpu # NVHPC Thrust (requires NVIDIA GPU + HPC SDK build) +--device gpu-omp # OpenMP target offload, NVIDIA and AMD GPUs +--device auto # GPU if available, else CPU +``` + +`sbd.available_backends()` reports what this install actually has (a static scan +— it does not import anything, so it is safe to call outside `mpirun`), and +`sbd.loaded_backends()` reports what the current process has pulled in. + +Within Python, backends can be selected per call — no re-initialization needed: + +```python +import sbd + +# No init() needed — auto-initializes on first call +result_cpu = sbd.tpb_diag(..., device='cpu') +result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' +# result_omp = sbd.tpb_diag(..., device='gpu-omp') # do NOT mix with 'cpu' — see below +``` + +**`'cpu'` and `'gpu-omp'` must not be used in the same process.** Both link the same +OpenMP runtime, and `_core_cpu` is built without offload support, so whichever loads +first initializes that runtime host-only. If it is the CPU backend, the OMP-offload +backend can no longer acquire a device and **silently runs its target regions on the +host**: correct energies, exit status 0, and the GPU sitting idle. There is no error to +catch, which is why it is worth knowing rather than discovering. `sbd.loaded_backends()` +reports what the current process has actually imported. + +`'cpu'` and `'gpu'` (Thrust) *can* share a process — Thrust does not route its device +work through OpenMP, so it has no equivalent interaction. + +## Available Test Data + +Paths below are written as the drivers use them, i.e. relative to a solver folder +such as `examples/tpb/`, which is how the drivers' own defaults are spelled. + +**H2O** (`../../vendor/sbd-upstream/data/h2o/`): `h2o-1em3` through `h2o-1em8` alpha determinant files. +**N2** (`../../vendor/sbd-upstream/data/n2/`): `1em3` through `1em7` and `3em4` through `3em7` alpha determinant files. + +Smaller thresholds = more determinants = higher accuracy. + +## Performance Tips + +**CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). + +**GPU:** One MPI rank per GPU — each rank is auto-assigned `gpu_id = rank % num_gpus`. +Use method 0 (matrix-free Davidson) for best GPU performance. + +**Do not set `OMP_NUM_THREADS=1` for GPU runs.** One rank per GPU is about device +ownership, not thread count, and a GPU build still does real work on the host: helper +construction is host-threaded in both solvers (`tpb/helper.h`, `gdb/helper.h`), and for +GDB the heatbath expansion and carryover selection have no device implementation at all +(`gdb/expansion.h`, `gdb/carryover.h` carry no `thrust::` code and are compiled in +regardless of backend). One thread per rank single-threads all of it. Divide the node's +physical cores among the ranks exactly as for a CPU run — e.g. 96 cores with 8 ranks is +`OMP_NUM_THREADS=12`. Expect GPU utilisation below 100% as a result; that is the host +phases, not a fault. + +## See Also + +- [Repository README](../../README.md) — Installation, API reference diff --git a/python/examples/count_dict_h2o.json b/examples/tpb/count_dict_h2o.json similarity index 100% rename from python/examples/count_dict_h2o.json rename to examples/tpb/count_dict_h2o.json diff --git a/python/examples/run_sbd_diag.py b/examples/tpb/run_sbd_diag.py similarity index 100% rename from python/examples/run_sbd_diag.py rename to examples/tpb/run_sbd_diag.py diff --git a/python/examples/run_sqd_enlarge_subspace_sbd.py b/examples/tpb/run_sqd_enlarge_subspace_sbd.py similarity index 100% rename from python/examples/run_sqd_enlarge_subspace_sbd.py rename to examples/tpb/run_sqd_enlarge_subspace_sbd.py diff --git a/python/examples/run_sqd_sbd.ipynb b/examples/tpb/run_sqd_sbd.ipynb similarity index 100% rename from python/examples/run_sqd_sbd.ipynb rename to examples/tpb/run_sqd_sbd.ipynb diff --git a/python/examples/run_sqd_sbd.py b/examples/tpb/run_sqd_sbd.py similarity index 100% rename from python/examples/run_sqd_sbd.py rename to examples/tpb/run_sqd_sbd.py diff --git a/python/__init__.py b/python/__init__.py index c2a683c..d74a2fe 100644 --- a/python/__init__.py +++ b/python/__init__.py @@ -465,7 +465,7 @@ def get_device_id(device=None): Uses the communicator rank, i.e. the same ``gpu_id = rank % num_gpus`` convention SBD's own diagonalization applies, documented in - ``python/examples/README.md``. + ``examples/tpb/README.md``. """ _ensure_initialized() return get_backend(device).planned_device_id(get_comm()) diff --git a/python/examples/README.md b/python/examples/README.md index 66cbc25..e78af36 100644 --- a/python/examples/README.md +++ b/python/examples/README.md @@ -1,394 +1,10 @@ -# SQD/SBD Python Examples +# Moved -Examples demonstrating SBD's capabilities for quantum chemistry calculations. +The examples now live at the top level, outside the installed package, grouped by basis +type: -## Overview +- [`examples/tpb/`](../../examples/tpb/README.md) — TPB (tensor-product basis) and the + SQD loops, i.e. everything that used to be in this directory. Its README also covers + backend selection, the bundled test data and threading. -- **Communication:** MPI for distributed computing -- **Backends:** CPU (host OpenMP, `--device cpu`), GPU (NVHPC Thrust, NVIDIA only, `--device gpu`) and GPU (OpenMP target offload, NVIDIA and AMD, `--device gpu-omp`), switchable at runtime via `device` parameter - -Replace `--device gpu` in all the examples below with `--device gpu-omp` if you use AMD GPUs. - -## Examples - -### 1. run_sbd_diag.py — Standalone SBD Diagonalization - -Runs a single TPB diagonalization from an FCIDUMP file and alpha determinant -file. No SQD loop, no Qiskit dependency. - -```bash -# H2O with 2 MPI ranks with CPU -mpirun -np 2 python -u run_sbd_diag.py \ - --device cpu \ - --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ - --adetfile ../../vendor/sbd-upstream/data/h2o/h2o-1em3-alpha.txt \ - --adet_comm_size 2 - -# N2 with GPU -mpirun -np 8 python -u run_sbd_diag.py \ - --device gpu \ - --fcidump ../../vendor/sbd-upstream/data/n2/fcidump.txt \ - --adetfile ../../vendor/sbd-upstream/data/n2/1em3-alpha.txt \ - --adet_comm_size 4 --bdet_comm_size 2 - -# Retrieve the 1-/2-particle RDMs and save them to a file -mpirun -np 2 python -u run_sbd_diag.py --rdm_output /tmp/h2o_rdms.npz -``` - -`--rdm_output` takes the file to save to (`/tmp/h2o_rdms.npz` above) and -writes **one** `.npz` file there holding both `rdm1` and `rdm2` together -(`data = np.load("/tmp/h2o_rdms.npz"); data["rdm1"]`, `data["rdm2"]`) — -unlike upstream SBD's own CLI, which writes two separate files -(`1pRDM.txt`/`2pRDM.txt`). It also prints `trace(rdm1)` and the natural -orbital occupations. Leaving it empty (the default) skips computing RDMs -entirely. - -By default beta determinants are derived from `--adetfile` alone (identical -to it, or a shuffled copy if `--shuffle` is set). `--symmetrize_spin 0` -loads `--adetfile` and `--bdetfile` as independent, genuinely distinct -alpha/beta determinant sets instead; `--bdetfile` is otherwise ignored -(with a warning) since symmetric mode always derives beta from alpha. - -**Key options:** `--device`, `--fcidump`, `--adetfile`, `--bdetfile`, -`--symmetrize_spin`, `--adet_comm_size`, `--bdet_comm_size`, -`--task_comm_size`, `--method`, `--tolerance`, `--iteration`, -`--rdm_output`. (These keep their unprefixed names here: this driver *is* -SBD. The SQD drivers prefix them `--sbd_*`.) Run `python run_sbd_diag.py ---help` for the full list. - -**Requirements:** `sbd`, `mpi4py` - -### 2. run_sqd_sbd.py — SQD Loop with SBD Solver - -Runs the self-consistent SQD workflow (qiskit-addon-sqd) using SBD as the -eigensolver backend. Supports two bitstring input modes: - -- `--counts FILE` — load bitstrings from a count_dict.json -- `--samples N` — generate N random bitstrings at the target Hamming weights - (default). A plumbing check only: random determinants give a random subspace, - so the energy is not meaningful. Use `--counts` for real results. - -```bash -# H2O with the bundled counts file (275 bitstrings -> ~ -76.236 Ha) -mpirun -np 8 python -u run_sqd_sbd.py \ - --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ - --counts count_dict_h2o.json \ - --device gpu \ - --adet_comm_size 4 --bdet_comm_size 2 - -# Custom system with random bitstrings -mpirun -np 8 python -u run_sqd_sbd.py \ - --fcidump /path/to/fci_dump.txt \ - --samples_per_batch 800 --num_batches 3 --max_iterations 10 \ - --device gpu \ - --adet_comm_size 4 --bdet_comm_size 2 -``` - -**count_dict.json format:** A JSON object mapping bitstrings to shot counts, as -produced by a quantum device or simulator. Each bitstring has length `2 × NORB` -and is laid out as **`[beta | alpha]`**: the first `NORB` bits are beta -(spin-down), the last `NORB` are alpha (spin-up), and within each half **orbital 0 -is the rightmost bit**. qiskit-addon-sqd postselects the last `NORB` bits on -`num_elec_a` and the first `NORB` on `num_elec_b`. - -For H2O (NORB=24, 5α+5β) the Hartree–Fock configuration — the five lowest -orbitals doubly occupied — is therefore `"0"*19 + "1"*5` in *both* halves: - -```json -{ - "000000000000000000011111000000000000000000011111": 16, - "010000000010001010000001010000000001000010100100": 12, - "000010001110000000000010001001000110000000000100": 8 -} -``` - -Bitstrings whose halves do not hold exactly `num_elec_a` / `num_elec_b` ones are -dropped by postselection, so a file of uniform-random strings yields nothing -usable — for H2O only `C(24,5)² / 4²⁴ ≈ 6e-6` of them qualify. - -[`count_dict_h2o.json`](./count_dict_h2o.json) in this directory is a ready-made -H2O example: 275 bitstrings taken from the vendored `h2o-1em3-alpha.txt` -determinant list, giving a 275 × 275 = 75,625-determinant subspace at -≈ -76.236 Ha. - -**Key options:** `--fcidump` (required), `--counts`, `--samples`, -`--samples_per_batch`, `--num_batches`, `--max_iterations`, `--device`, -MPI decomposition flags, and the SQD tolerances `--energy_tol` / -`--occupancies_tol` / `--sqd_carryover_threshold`. Inner-solver flags are prefixed -(`--sbd_eps`, `--sbd_max_it`, `--sbd_method`, ...) and have sensible defaults; the -old unprefixed spellings still work. Run `python run_sqd_sbd.py --help` for the -full list, which is grouped by layer. - -**Requirements:** see [Integration with qiskit-addon-sqd](../../README.md#integration-with-qiskit-addon-sqd) in the Python Bindings README (`pyscf`, `qiskit`, `qiskit-addon-sqd`). - -See [SQD Parameters](#sqd-parameters) below for the full reference, grouped by SQD loop / SBD solver / MPI grid / checkpointing. - -### 3. run_sqd_enlarge_subspace_sbd.py — SQD that grows its own subspace - -Same self-consistent SQD loop as `run_sqd_sbd.py` above (sampling, -configuration recovery, SBD as the solver), but with one addition: between -rounds, it expands the dominant determinant pairs from the just-solved -wavefunction via qiskit-addon-sqd's own `enlarge_batch_from_transitions` -(same-spin single-electron excitations, both alpha and beta), and feeds the -result forward as the next round's `include_configurations`. Concretely, -it calls `diagonalize_fermionic_hamiltonian` with `max_iterations=1` itself, -in its own outer Python loop, rather than delegating the whole -multi-iteration loop to one call the way `run_sqd_sbd.py` does -- that's -what makes injecting a step between rounds possible. The loop stops when -either the expanded set adds nothing new, or the energy and occupancies -both stop moving (`--energy_tol`/`--occupancies_tol`) -- `--max_iterations` -is a safety cap, not the expected stopping mechanism. - -```bash -mpirun -np 8 python -u run_sqd_enlarge_subspace_sbd.py \ - --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ - --counts count_dict_h2o.json \ - --device gpu \ - --adet_comm_size 4 --bdet_comm_size 2 --enlarge_threshold 1e-4 -``` - -**A note on JAX and GPUs.** The expansion step above is JAX. Since XLA reserves 75% of a device -on first use, one rank claims it and the rest fail with `RESOURCE_EXHAUSTED: CUDA_ERROR_OUT_OF_MEMORY`. -On a GPU backend this driver therefore does two things for you, so the command above needs no extra environment: - -- assigns each rank its own device, using `sbd.get_device_id()` so JAX lands - on the same card SBD already selected for that rank; -- sets `XLA_PYTHON_CLIENT_PREALLOCATE=false`, so XLA takes memory on demand - instead of reserving 75% of the card away from SBD. Without this, SBD's - Davidson basis can run out of room at large `--max_dim` and fail inside - `tpb_diag` with `std::bad_alloc` -- a memory error that reads as SBD's - fault but is caused by JAX's reservation. - -`XLA_PYTHON_CLIENT_PREALLOCATE` is left alone if you set it yourself, and -`JAX_PLATFORMS=cpu` still forces the expansion onto CPU. The device assignment -always runs, and stays correct if you pin one GPU per rank yourself: with a -single card visible the count is 1, so every rank resolves to device 0 -- its -own. - -Same bundled 275-bitstring H2O pool as `run_sqd_sbd.py`'s own example above: -plain SQD reaches **≈ -76.236 Ha** and stops there; this driver keeps going -past that fixed pool on its own and converges to **-76.2421767512 Ha**. - -See [SQD Parameters](#sqd-parameters) below for the flags it shares with -`run_sqd_sbd.py` and the ones that differ (`--enlarge_threshold` in place -of `--sqd_carryover_threshold`, and `--max_dim`'s risk profile is sharper -here). - -### 4. run_sqd_sbd.ipynb — Jupyter walkthrough (serial) - -Interactive single-rank companion to `run_sqd_sbd.py`. Same SQD self-consistent -loop on h2o, but inside a Jupyter kernel (`MPI.COMM_WORLD` size 1). Uses the -bundled [`count_dict_h2o.json`](./count_dict_h2o.json) (275 bitstrings → 75,625 -determinants) and reaches ≈ −76.236 Ha in a few seconds on CPU. - -```bash -pytest --nbmake run_sqd_sbd.ipynb # what CI runs; needs the nbtest extra -# or open it in JupyterLab and step through the cells -``` - -## SQD Parameters - -Reference for every flag `run_sqd_sbd.py` accepts, grouped the way `--help` -groups them: SQD loop, SBD solver, MPI grid, checkpointing. - -**How each iteration builds its subspace.** SQD samples bitstrings from a -quantum device, repairs the noisy ones against an orbital-occupancy estimate -(**configuration recovery**), subsamples them into batches, and diagonalizes -each batch. What makes it a *loop* is that two results feed back into the next -iteration. Three sources, concatenated in this priority order inside -qiskit-addon-sqd's `diagonalize_fermionic_hamiltonian` (`fermion.py`): - -``` -strs_a = include_a ++ carryover_strings_a ++ samples_a then dedupe, truncate to max_dim, sort -``` - -1. **`include_a`** — configurations passed as `include_configurations`. Static: - fixed before the loop, present every iteration, never updated. -2. **`carryover_strings_a`** — from the *previous* iteration's wavefunction. Every - determinant whose `|coefficient|` is at least `--sqd_carryover_threshold` - survives, ranked by `|c|^2`. -3. **`samples_a`** — drawn fresh this iteration, sorted by marginal probability. - -The order matters when `max_dim` is set: `include` and `carryover` are kept ahead of -fresh samples, so if those two already fill the cap, this iteration's new samples are -truncated away entirely. - -The samples are not re-used raw counts. Each iteration re-runs configuration -recovery from the *original* bitstrings using the occupancies from the previous -iteration's best batch (`_prepare_ci_strings` in `fermion.py`), then -subsamples. Recovery is **not -cumulative** — it always re-derives from the raw samples, just with a better -occupancy estimate each time. On iteration 1 there are no occupancies yet, so the -raw samples are only filtered by electron count (Hamming-weight postselection). - -So exactly two things flow from iteration N to N+1, and neither is a tolerance: -the **average orbital occupancies** (into recovery, source 3) and the -**wavefunction amplitudes** (into carryover, source 2). - -### SQD loop parameters - -Shared by both `run_sqd_sbd.py` and `run_sqd_enlarge_subspace_sbd.py` except -where noted. **`--max_dim` is the one that most needs attention**: it has no -universally safe default (see below), and in `run_sqd_enlarge_subspace_sbd.py` -leaving it unset is riskier still, since each round's subspace can grow from -the previous one rather than being resampled at a fixed size — the driver -prints an OOM warning when it detects this. - -*Shapes the subspace — changes the numbers you compute:* - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--counts FILE` | Load hardware bitstrings from a JSON file (use this or `--samples`) | none — falls back to `--samples` if omitted | -| `--samples N` | Generate N random bitstrings at the target Hamming weights; plumbing check only, energy not meaningful | `3000` (only used when `--counts` is omitted) | -| `--samples_per_batch` | Dominant control on subspace dimension. With `--symmetrize_spin 1` the alpha and beta string sets are merged, so the subspace is up to `(2N)^2`, not `N^2` | `3000` | -| `--symmetrize_spin` | `1` (default): merge the alpha and beta string pools every iteration, forcing `ci_strs_a == ci_strs_b`. SBD itself supports distinct alpha/beta determinant sets — this is purely a qiskit-addon-sqd loop-layer setting. `0`: sample and carry over alpha and beta independently, allowing them to differ | `1` | -| `--num_batches` | Independent subsamples per iteration; occupancies are averaged across them | `1` (`run_sqd_enlarge_subspace_sbd.py`) / `3` (`run_sqd_sbd.py`) | -| `--sqd_carryover_threshold` | `run_sqd_sbd.py` only. `\|coefficient\|` cutoff for carrying a determinant into the next iteration's sample pool. **Lower it to carry more** | `1e-4` | -| `--enlarge_threshold` | `run_sqd_enlarge_subspace_sbd.py` only — the analogous "carry more" knob for that driver, but structurally different: it gates which *pairs* get expanded into single excitations via `enlarge_batch_from_transitions`, not which determinants survive into resampling. **Lower it to expand more pairs per round** | `1e-4` | -| `--max_dim` | **Critical.** Cap on strings per spin sector, so the subspace cannot exceed `max_dim^2`. The main brake on runaway cost — no fixed value is safe for every system, since the right cap depends on available memory and orbital count. Start from a value known to work at a similar orbital count (e.g. `15000` was used for a 45-orbital system) and adjust down if you see an OOM | unset (no cap) | -| `--include_hf` | Force the single Slater determinant with the lowest `num_elec_a`/`num_elec_b` orbital indices occupied into `include_configurations`, every iteration. Cheap correctness check: that determinant's own diagonal energy is an exact lower bound on what a subspace containing it can do — if forcing it in moves the result, the sampled pool was missing it (and probably its low-excitation neighbors too) | off | - -*Decides when to stop — changes nothing about the subspace:* - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--max_iterations` | Hard cap on loop iterations (not the inner `--sbd_max_it`). In `run_sqd_enlarge_subspace_sbd.py` this is a safety cap only — the loop normally stops earlier, once a round adds no new determinants or both tolerances below are met | `30` (`run_sqd_enlarge_subspace_sbd.py`) / `5` (`run_sqd_sbd.py`) | -| `--energy_tol` | Iteration-to-iteration change in energy | `1e-8` | -| `--occupancies_tol` | Largest change in any single orbital occupancy — an infinity norm, not an average | `1e-5` | - -**Both stopping criteria must hold in the same iteration.** `fermion.py`'s -convergence check combines the energy-change test and the occupancy-change test -with a logical `and`, so the loop only stops once both are satisfied at once — -not whichever one happens first. A run that reaches -`--max_iterations` may be converged in energy while one stubborn orbital's -occupancy is still moving, and loosening only one tolerance will not stop it. -Watch the per-batch energies: while they still disagree, the loop has not -converged regardless of what the total says. - -### SBD solver parameters - -Per diagonalization, not per loop: - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--sbd_eps` | Davidson stop: **norm of the residual vector**, not an energy. Error in the energy goes roughly as `\|R\|^2/gap`, so this already implies far better energy accuracy than `--energy_tol` asks for. Tighten it for a near-degenerate system | `1e-5` | -| `--sbd_max_it` | Cap on Davidson iterations. Reaching it before `--sbd_eps` returns a partially converged vector **with no warning** — watch the `tol=` values SBD prints, and cross-batch agreement | `10` | -| `--sbd_max_nb` | Davidson basis vectors (block size). Peak memory during Davidson grows with how many sub-iterations it actually needs, not just `--max_dim` — a run that succeeds for several iterations at a fixed `dim` can still later need more basis vectors and run out of memory even though the subspace itself did not grow. Lowering this trades some convergence robustness for a lower memory ceiling | `10` | -| `--sbd_method` | 0=Davidson, 1=Davidson+Ham, 2=Lanczos, 3=Lanczos+Ham | `0` | -| `--sbd_use_precalculated_dets` | Thrust only. `1` precomputes a determinant index for **every** (α,β) pair — the whole subspace, on the GPU. `0` uses per-thread storage: slower per matvec, far less memory | `1` | -| `--sbd_max_memory_gb_for_determinants` | Thrust only, and **only consulted when `--sbd_use_precalculated_dets 0`** (`mult_thrust.h:257-273`). Caps the per-thread buffer in GB | `-1` (uncapped) | - -For reference, upstream's own `TPB_SBD` struct defaults are looser still (`max_it=1`, -`eps=1e-4`), and `run_sbd_diag.py` uses `eps=1e-3`. On the h2o counts case, -`eps=1e-5` and `eps=1e-8` give the same energy to ten decimal places. - -### MPI grid parameters - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--adet_comm_size` | Ranks spanning the alpha-determinant dimension | `1` | -| `--bdet_comm_size` | Ranks spanning the beta-determinant dimension | `1` | -| `--task_comm_size` | Ranks spanning task-level parallelism | `1` | - -All ranks diagonalize each batch together, then move to the next batch -sequentially. See [MPI Decomposition](#mpi-decomposition) below for the full 4D -grid (a fourth, *derived* dimension — `helper` — is not set directly). - -### Checkpointing parameters - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--checkpoint_path` | Write `ci_strs_a`/`ci_strs_b`/`orbital_occupancies`/energy to this path as JSON text (rank 0 only), every `--checkpoint_frequency` iterations. **Must be visible under the same path from every rank** — `--resume_from` has no rank-0-reads-then-broadcasts step, every rank opens this path itself | unset | -| `--checkpoint_frequency` | Write every this many iterations, plus always on the last one regardless of alignment | `1` (every iteration) | -| `--resume_from` | Seed a new run's `include_configurations`/`initial_occupancies` from a previous `--checkpoint_path`'s last recorded iteration | unset | - -What's preserved is which determinants (`ci_strs_a`/`ci_strs_b`, as plain -integers — independent of MPI decomposition, since they carry no rank/grid -information) and the derived per-orbital occupancy averages. **Not** the -wavefunction amplitudes matrix itself (`dim_a × dim_b` floats — at large -`--max_dim` this alone would dwarf everything else; neither consuming parameter -needs it). - -Not a bit-identical continuation: RNG state is fresh in the new process, and -every string from the resumed iteration becomes a **permanent** include for -every iteration of the new run — unlike a true single-process continuation, -where `--sqd_carryover_threshold` would keep pruning low-weight determinants -each iteration. A resumed run is seeded richer than an actual continuation -would have been at that point, not identical to one. - -## MPI Decomposition - -Total MPI ranks must be a **multiple** of -`task_comm_size × adet_comm_size × bdet_comm_size` — not equal to it. SBD splits -the ranks you asked for across those three dimensions and puts whatever remains -into a fourth, "helper" dimension, computed as -`ranks / (task_comm_size × adet_comm_size × bdet_comm_size)`. - -So 8 ranks with `--adet_comm_size 2 --bdet_comm_size 2` is valid: the grid is -`1 × 2 × 2` and the helper dimension absorbs the remaining factor of 2. - -When using more than one rank, specify at least `--adet_comm_size`. Examples: - -| Ranks | Decomposition | Helper | -|-------|---------------|--------| -| 1 | default (all = 1) | 1 | -| 2 | `--adet_comm_size 2` | 1 | -| 4 | `--adet_comm_size 2 --bdet_comm_size 2` | 1 | -| 8 | `--adet_comm_size 2 --bdet_comm_size 2` | 2 | -| 8 | `--adet_comm_size 2 --bdet_comm_size 2 --task_comm_size 2` | 1 | - -**GDB** (`gdb_diag`) decomposes differently: `t_comm_size × b_comm_size × helper`, -with its own field names rather than TPB's. It is not exercised by these examples -or by the test suite, so its decomposition is unvalidated and is deliberately not -documented further here. - -## Backend Selection - -Every backend the toolchain supported was compiled into this one install, and -each is imported **lazily, on first use** — normally one per process. Select -per-call via `--device`: - -```bash ---device cpu # host OpenMP (default) ---device gpu # NVHPC Thrust (requires NVIDIA GPU + HPC SDK build) ---device gpu-omp # OpenMP target offload, NVIDIA and AMD GPUs ---device auto # GPU if available, else CPU -``` - -`sbd.available_backends()` reports what this install actually has (a static scan -— it does not import anything, so it is safe to call outside `mpirun`), and -`sbd.loaded_backends()` reports what the current process has pulled in. - -Within Python, backends can be selected per call — no re-initialization needed: - -```python -import sbd - -# No init() needed — auto-initializes on first call -result_cpu = sbd.tpb_diag(..., device='cpu') -result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' -# result_omp = sbd.tpb_diag(..., device='gpu-omp') # NOT in the same process as 'cpu' -``` - -## Available Test Data - -**H2O** (`../../vendor/sbd-upstream/data/h2o/`): `h2o-1em3` through `h2o-1em8` alpha determinant files. -**N2** (`../../vendor/sbd-upstream/data/n2/`): `1em3` through `1em7` and `3em4` through `3em7` alpha determinant files. - -Smaller thresholds = more determinants = higher accuracy. - -## Expected Results - -- **H2O**: ground state energy ≈ **-76.236 Hartree** -- **N2**: ground state energy ≈ **-109.042 Hartree** (with 1e-3 dets) - -## Performance Tips - -**CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). - -**GPU:** One MPI rank per GPU, `OMP_NUM_THREADS=1`. Each rank auto-assigned: `gpu_id = rank % num_gpus`. Use method 0 (matrix-free Davidson) for best GPU performance. - -## See Also - -- [Python Bindings README](../../README.md) — Installation, API reference -- [Upstream SBD library](https://github.com/r-ccs-cms/sbd) — C++ library overview +This file is a signpost for older links and can be removed once they have aged out. diff --git a/setup.py b/setup.py index 63bff45..394d1a9 100644 --- a/setup.py +++ b/setup.py @@ -473,19 +473,42 @@ def _homebrew_prefix(): return None +# Where nvc++ sits relative to whichever directory the environment points at. +# NVHPC installs the compiler in /compilers/bin, so probing only +# /bin misses the value NVIDIA itself hands out -- see NVHPC_VARS below. +_NVHPC_BIN_RELATIVE = ('bin', os.path.join('compilers', 'bin')) + +# NVHPC_HOME is this package's own variable and is documented in INSTALL.md as +# the compilers directory. NVHPC_ROOT is set by NVIDIA's shipped modulefile, to +# the version root .../Linux_x86_64/ -- that file does +# setenv NVHPC_ROOT $nvhome/$target/$version +# prepend-path PATH $nvcompdir/bin +# so the compiler is one level down from NVHPC_ROOT. Accept both variables, and +# under each accept both layouts, so that the compilers directory, the version +# root, and a copy of $NVHPC_ROOT all work rather than silently yielding a +# CPU-only build. +_NVHPC_VARS = ('NVHPC_HOME', 'NVHPC_ROOT') + + def find_nvidia_hpc_sdk(): - nvhpc_home = os.environ.get('NVHPC_HOME', None) - if nvhpc_home: - nvcxx_path = os.path.join(nvhpc_home, 'bin', 'nvc++') - if os.path.exists(nvcxx_path): - print(f"Found NVIDIA HPC SDK at: {nvhpc_home}") - nvhpc_bin = os.path.join(nvhpc_home, 'bin') - current_path = os.environ.get('PATH', '') - if nvhpc_bin not in current_path: - os.environ['PATH'] = f"{nvhpc_bin}:{current_path}" - return nvcxx_path, True - else: - print(f"Warning: NVHPC_HOME set to {nvhpc_home} but nvc++ not found") + """Locate nvc++: the NVHPC_* variables first, then PATH.""" + for var in _NVHPC_VARS: + root = os.environ.get(var) + if not root: + continue + for relative in _NVHPC_BIN_RELATIVE: + nvcxx_path = os.path.join(root, relative, 'nvc++') + if os.path.exists(nvcxx_path): + print(f"Found NVIDIA HPC SDK at: {nvcxx_path} (via {var})") + nvhpc_bin = os.path.dirname(nvcxx_path) + current_path = os.environ.get('PATH', '') + # Compare path entries, not substrings: a directory whose name + # happens to be contained in some other entry is not on PATH. + if nvhpc_bin not in current_path.split(os.pathsep): + os.environ['PATH'] = f"{nvhpc_bin}{os.pathsep}{current_path}" + return nvcxx_path, True + tried = ', '.join(os.path.join(root, r, 'nvc++') for r in _NVHPC_BIN_RELATIVE) + print(f"Warning: {var} is set but no nvc++ found. Tried: {tried}") import shutil nvcxx_path = shutil.which('nvc++') if nvcxx_path: diff --git a/test/conftest.py b/test/conftest.py index 159a06a..49040d2 100644 --- a/test/conftest.py +++ b/test/conftest.py @@ -28,9 +28,7 @@ # So a missing file means a broken checkout or a move that did not update this file, and # the tests should fail and say so rather than skip and report success. DATA_DIR = Path(__file__).resolve().parents[1] / "vendor" / "sbd-upstream" / "data" -COUNTS_PATH = ( - Path(__file__).resolve().parents[1] / "python" / "examples" / "count_dict_h2o.json" -) +COUNTS_PATH = Path(__file__).resolve().parents[1] / "examples" / "tpb" / "count_dict_h2o.json" # Slow tests are opt-in through a command-line flag rather than excluded by default, diff --git a/tox.ini b/tox.ini index dd3ecb9..0f98e64 100644 --- a/tox.ini +++ b/tox.ini @@ -69,7 +69,7 @@ extras = nbtest notebook-dependencies commands = - pytest --nbmake --nbmake-timeout=3000 {posargs} python/examples/ + pytest --nbmake --nbmake-timeout=3000 {posargs} examples/ [testenv:slow] # The reference cases grow by roughly an order of magnitude in determinant count per