From 103c5339def6e50ae67915802f46e4a54f4eb176 Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 12:49:53 -0400 Subject: [PATCH 01/10] examples: move out of the package to a top-level examples/ directory Pure reorganization, no behaviour change, so that a later PR adding a second solver's drivers is additive rather than a rename plus a feature. Runnable MPI scripts were living inside the installed package: they shipped in the wheel, sat beside __init__.py and bindings.cpp, and a path like python/examples/run_sbd_diag.py reads as package internals. They move to a top-level examples/, which is the common convention for runnable scripts, and are grouped by basis type -- examples/tpb/ -- since the solvers take different subspaces and decompose over MPI differently. (The sibling qiskit-addon-sqd has no examples/ only because its tutorials are notebooks rendered into Sphinx docs; ours are CLI drivers.) examples/tpb/ is the same depth as the old location, so every ../../vendor/sbd-upstream/... path inside the drivers keeps working untouched. What is package-level rather than example-level -- backend selection, the --device table, bundled test data, performance notes -- moves to examples/README.md, which the solver README links to instead of restating. The GDB decomposition paragraph goes there too so nothing is lost; a later PR relocates and expands it. python/examples/README.md stays behind as a signpost, since git mv removes the directory outright and links to python/examples/... would otherwise 404. It can be deleted once those links age out. The top README now names the folders and defers to their READMEs instead of listing every example file, which is the part that rots whenever a driver is added. Reference updates: MANIFEST.in, tox.ini's notebook env, python/__init__.py's citation of the rank-to-device rule. Two small fixes ride along: MANIFEST.in gained *.json so count_dict_h2o.json -- which run_sqd_sbd.py defaults to -- now reaches the sdist, and the bundled-data paths in the shared README are stated relative to a solver folder, matching how the drivers' own defaults are spelled. Verified: run_sbd_diag.py runs from examples/tpb/ resolving its default vendored paths (-76.2359465468); the sdist contains examples/, examples/tpb/ with all six files, and the signpost; no reference to python/examples survives except the MANIFEST line that ships the signpost. Co-Authored-By: Claude Opus 5 (1M context) --- MANIFEST.in | 3 +- README.md | 24 +- examples/README.md | 75 ++++ examples/tpb/README.md | 348 +++++++++++++++ .../tpb}/count_dict_h2o.json | 0 .../examples => examples/tpb}/run_sbd_diag.py | 0 .../tpb}/run_sqd_enlarge_subspace_sbd.py | 0 .../tpb}/run_sqd_sbd.ipynb | 0 .../examples => examples/tpb}/run_sqd_sbd.py | 0 python/__init__.py | 2 +- python/examples/README.md | 398 +----------------- tox.ini | 2 +- 12 files changed, 446 insertions(+), 406 deletions(-) create mode 100644 examples/README.md create mode 100644 examples/tpb/README.md rename {python/examples => examples/tpb}/count_dict_h2o.json (100%) rename {python/examples => examples/tpb}/run_sbd_diag.py (100%) rename {python/examples => examples/tpb}/run_sqd_enlarge_subspace_sbd.py (100%) rename {python/examples => examples/tpb}/run_sqd_sbd.ipynb (100%) rename {python/examples => examples/tpb}/run_sqd_sbd.py (100%) diff --git a/MANIFEST.in b/MANIFEST.in index f8bddd5..4d49a21 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -16,4 +16,5 @@ recursive-include vendor/sbd-upstream/include *.h include vendor/sbd-upstream/LICENSE.txt # Example scripts and notebook, referenced by the README. -recursive-include python/examples *.py *.ipynb *.md +recursive-include examples *.py *.ipynb *.md *.json +include python/examples/README.md diff --git a/README.md b/README.md index 5664190..6d41dac 100644 --- a/README.md +++ b/README.md @@ -181,7 +181,7 @@ MPICC=$MPI_HOME/bin/mpicc python -m pip install --no-binary=mpi4py --no-cache-di # confirm which MPI mpi4py uses -- setup.py builds against exactly this python -c "from mpi4py import MPI; print(MPI.Get_library_version())" -# only for the SQD examples (python/examples/run_sqd_sbd.py and .ipynb) +# only for the SQD examples (examples/tpb/run_sqd_sbd.py and .ipynb) pip install "qiskit-addon-sqd>=0.13.1" # install sbd-eigensolver @@ -213,14 +213,14 @@ The Thrust backend is stamped too (`cuda:cc90`); the CPU backend reports `None`. ## Examples -Located in `python/examples/`: +Located in [`examples/`](examples/README.md), organized by basis type since the +solvers take different subspaces and decompose over MPI differently. Each folder's +README is the authoritative list of what it contains and how to run it. -- [`run_sbd_diag.py`](python/examples/run_sbd_diag.py) — Standalone TPB diagonalization (no Qiskit dependency) -- [`run_sqd_sbd.ipynb`](python/examples/run_sqd_sbd.ipynb) — Jupyter Notebook SQD loop with SBD solver (random or hardware bitstrings) -- [`run_sqd_sbd.py`](python/examples/run_sqd_sbd.py) — SQD loop with SBD solver (random or hardware bitstrings) -- [`run_sqd_enlarge_subspace_sbd.py`](python/examples/run_sqd_enlarge_subspace_sbd.py) — SQD that also grows its own subspace between rounds via single excitations - -See [python/examples/README.md](python/examples/README.md) for usage details. +- [`examples/tpb/`](examples/tpb/README.md) — tensor-product basis: standalone TPB + diagonalization, the SQD loops, and the subspace-enlargement driver. +- [`examples/README.md`](examples/README.md) — backend selection, `--device` values, + bundled test data and performance notes, shared by all examples. ## Integration with qiskit-addon-sqd @@ -255,9 +255,9 @@ result = diagonalize_fermionic_hamiltonian( ) ``` -See [SQD Parameters](python/examples/README.md#sqd-parameters) for how each +See [SQD Parameters](examples/tpb/README.md#sqd-parameters) for how each parameter feeds the loop, and -[python/examples/run_sqd_sbd.py](python/examples/run_sqd_sbd.py) for a +[examples/tpb/run_sqd_sbd.py](examples/tpb/run_sqd_sbd.py) for a complete example. qiskit-addon-sqd is the orchestrator in that recipe: it owns the loop @@ -267,7 +267,7 @@ say in how the subspace grows between iterations. ### SQD with subspace enlargement -[`run_sqd_enlarge_subspace_sbd.py`](python/examples/run_sqd_enlarge_subspace_sbd.py) +[`run_sqd_enlarge_subspace_sbd.py`](examples/tpb/run_sqd_enlarge_subspace_sbd.py) builds on the same recipe, but grows its own subspace between rounds: after each solve, it expands the dominant determinant pairs via qiskit-addon-sqd's own `enlarge_batch_from_transitions` (same-spin single excitations, both @@ -277,7 +277,7 @@ alpha and beta) and feeds the result forward as the next round's own outer Python loop, rather than delegating the whole multi-iteration loop to one call — that is what makes injecting a step between rounds possible. -On the bundled H2O pool ([`count_dict_h2o.json`](python/examples/count_dict_h2o.json), +On the bundled H2O pool ([`count_dict_h2o.json`](examples/tpb/count_dict_h2o.json), 275 bitstrings), plain SQD reaches ≈ -76.236 Ha and stops there; this driver keeps going past that fixed pool on its own and converges to **-76.2421767512 Ha**. diff --git a/examples/README.md b/examples/README.md new file mode 100644 index 0000000..8835447 --- /dev/null +++ b/examples/README.md @@ -0,0 +1,75 @@ +# SBD examples + +Runtime behaviour shared by every example: which backend you get, what data is +bundled, and how to spend threads and GPUs. Solver-specific usage lives with the +examples themselves: + +- [`tpb/`](tpb/README.md) — TPB (tensor-product basis) and the SQD loops built on + it. The subspace is the Cartesian product of an alpha and a beta determinant list. + +- **Communication:** MPI for distributed computing +- **Backends:** CPU (host OpenMP, `--device cpu`), GPU (NVHPC Thrust, NVIDIA only, + `--device gpu`) and GPU (OpenMP target offload, NVIDIA and AMD, `--device + gpu-omp`), switchable at runtime via the `device` parameter + +Replace `--device gpu` with `--device gpu-omp` if you use AMD GPUs. + +## Backend Selection + +Every backend the toolchain supported was compiled into this one install, and +each is imported **lazily, on first use** — normally one per process. Select +per-call via `--device`: + +```bash +--device cpu # host OpenMP (default) +--device gpu # NVHPC Thrust (requires NVIDIA GPU + HPC SDK build) +--device gpu-omp # OpenMP target offload, NVIDIA and AMD GPUs +--device auto # GPU if available, else CPU +``` + +`sbd.available_backends()` reports what this install actually has (a static scan +— it does not import anything, so it is safe to call outside `mpirun`), and +`sbd.loaded_backends()` reports what the current process has pulled in. + +Within Python, backends can be selected per call — no re-initialization needed: + +```python +import sbd + +# No init() needed — auto-initializes on first call +result_cpu = sbd.tpb_diag(..., device='cpu') +result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' +# result_omp = sbd.tpb_diag(..., device='gpu-omp') # NOT in the same process as 'cpu' +``` + +## Available Test Data + +Paths below are written as the drivers use them, i.e. relative to a solver folder +such as `examples/tpb/`, which is how the drivers' own defaults are spelled. + +**H2O** (`../../vendor/sbd-upstream/data/h2o/`): `h2o-1em3` through `h2o-1em8` alpha determinant files. +**N2** (`../../vendor/sbd-upstream/data/n2/`): `1em3` through `1em7` and `3em4` through `3em7` alpha determinant files. + +Smaller thresholds = more determinants = higher accuracy. + +## Performance Tips + +**CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). + +**GPU:** One MPI rank per GPU, `OMP_NUM_THREADS=1`. Each rank auto-assigned: `gpu_id = rank % num_gpus`. Use method 0 (matrix-free Davidson) for best GPU performance. + +## GDB (general determinant basis) + +`gdb_diag` spans the subspace with an explicit determinant list rather than the +Cartesian product TPB uses, and decomposes as `t_comm_size × b_comm_size × helper` +with its own field names. It has no example driver yet, so its decomposition is not +documented further here. + +`b_comm_size` must currently be 1: every rank passes the whole determinant list, so +splitting the basis communicator would have each rank diagonalize the full subspace +while believing it held a shard. + +## See Also + +- [Repository README](../README.md) — Installation, API reference +- [Upstream SBD library](https://github.com/r-ccs-cms/sbd) — C++ library overview diff --git a/examples/tpb/README.md b/examples/tpb/README.md new file mode 100644 index 0000000..5e0d15a --- /dev/null +++ b/examples/tpb/README.md @@ -0,0 +1,348 @@ +# TPB examples — SQD/SBD with a tensor-product basis + +Examples demonstrating SBD's capabilities for quantum chemistry calculations. + +## Overview + +These examples all use TPB, where the subspace is the Cartesian product of an +alpha and a beta determinant list. For backend selection, `--device` values and +the bundled test data, see [`../README.md`](../README.md). + +## Examples + +### 1. run_sbd_diag.py — Standalone SBD Diagonalization + +Runs a single TPB diagonalization from an FCIDUMP file and alpha determinant +file. No SQD loop, no Qiskit dependency. + +```bash +# H2O with 2 MPI ranks with CPU +mpirun -np 2 python -u run_sbd_diag.py \ + --device cpu \ + --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ + --adetfile ../../vendor/sbd-upstream/data/h2o/h2o-1em3-alpha.txt \ + --adet_comm_size 2 + +# N2 with GPU +mpirun -np 8 python -u run_sbd_diag.py \ + --device gpu \ + --fcidump ../../vendor/sbd-upstream/data/n2/fcidump.txt \ + --adetfile ../../vendor/sbd-upstream/data/n2/1em3-alpha.txt \ + --adet_comm_size 4 --bdet_comm_size 2 + +# Retrieve the 1-/2-particle RDMs and save them to a file +mpirun -np 2 python -u run_sbd_diag.py --rdm_output /tmp/h2o_rdms.npz +``` + +`--rdm_output` takes the file to save to (`/tmp/h2o_rdms.npz` above) and +writes **one** `.npz` file there holding both `rdm1` and `rdm2` together +(`data = np.load("/tmp/h2o_rdms.npz"); data["rdm1"]`, `data["rdm2"]`) — +unlike upstream SBD's own CLI, which writes two separate files +(`1pRDM.txt`/`2pRDM.txt`). It also prints `trace(rdm1)` and the natural +orbital occupations. Leaving it empty (the default) skips computing RDMs +entirely. + +By default beta determinants are derived from `--adetfile` alone (identical +to it, or a shuffled copy if `--shuffle` is set). `--symmetrize_spin 0` +loads `--adetfile` and `--bdetfile` as independent, genuinely distinct +alpha/beta determinant sets instead; `--bdetfile` is otherwise ignored +(with a warning) since symmetric mode always derives beta from alpha. + +**Key options:** `--device`, `--fcidump`, `--adetfile`, `--bdetfile`, +`--symmetrize_spin`, `--adet_comm_size`, `--bdet_comm_size`, +`--task_comm_size`, `--method`, `--tolerance`, `--iteration`, +`--rdm_output`. (These keep their unprefixed names here: this driver *is* +SBD. The SQD drivers prefix them `--sbd_*`.) Run `python run_sbd_diag.py +--help` for the full list. + +**Requirements:** `sbd`, `mpi4py` + +### 2. run_sqd_sbd.py — SQD Loop with SBD Solver + +Runs the self-consistent SQD workflow (qiskit-addon-sqd) using SBD as the +eigensolver backend. Supports two bitstring input modes: + +- `--counts FILE` — load bitstrings from a count_dict.json +- `--samples N` — generate N random bitstrings at the target Hamming weights + (default). A plumbing check only: random determinants give a random subspace, + so the energy is not meaningful. Use `--counts` for real results. + +```bash +# H2O with the bundled counts file (275 bitstrings -> ~ -76.236 Ha) +mpirun -np 8 python -u run_sqd_sbd.py \ + --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ + --counts count_dict_h2o.json \ + --device gpu \ + --adet_comm_size 4 --bdet_comm_size 2 + +# Custom system with random bitstrings +mpirun -np 8 python -u run_sqd_sbd.py \ + --fcidump /path/to/fci_dump.txt \ + --samples_per_batch 800 --num_batches 3 --max_iterations 10 \ + --device gpu \ + --adet_comm_size 4 --bdet_comm_size 2 +``` + +**count_dict.json format:** A JSON object mapping bitstrings to shot counts, as +produced by a quantum device or simulator. Each bitstring has length `2 × NORB` +and is laid out as **`[beta | alpha]`**: the first `NORB` bits are beta +(spin-down), the last `NORB` are alpha (spin-up), and within each half **orbital 0 +is the rightmost bit**. qiskit-addon-sqd postselects the last `NORB` bits on +`num_elec_a` and the first `NORB` on `num_elec_b`. + +For H2O (NORB=24, 5α+5β) the Hartree–Fock configuration — the five lowest +orbitals doubly occupied — is therefore `"0"*19 + "1"*5` in *both* halves: + +```json +{ + "000000000000000000011111000000000000000000011111": 16, + "010000000010001010000001010000000001000010100100": 12, + "000010001110000000000010001001000110000000000100": 8 +} +``` + +Bitstrings whose halves do not hold exactly `num_elec_a` / `num_elec_b` ones are +dropped by postselection, so a file of uniform-random strings yields nothing +usable — for H2O only `C(24,5)² / 4²⁴ ≈ 6e-6` of them qualify. + +[`count_dict_h2o.json`](./count_dict_h2o.json) in this directory is a ready-made +H2O example: 275 bitstrings taken from the vendored `h2o-1em3-alpha.txt` +determinant list, giving a 275 × 275 = 75,625-determinant subspace at +≈ -76.236 Ha. + +**Key options:** `--fcidump` (required), `--counts`, `--samples`, +`--samples_per_batch`, `--num_batches`, `--max_iterations`, `--device`, +MPI decomposition flags, and the SQD tolerances `--energy_tol` / +`--occupancies_tol` / `--sqd_carryover_threshold`. Inner-solver flags are prefixed +(`--sbd_eps`, `--sbd_max_it`, `--sbd_method`, ...) and have sensible defaults; the +old unprefixed spellings still work. Run `python run_sqd_sbd.py --help` for the +full list, which is grouped by layer. + +**Requirements:** see [Integration with qiskit-addon-sqd](../../README.md#integration-with-qiskit-addon-sqd) in the Python Bindings README (`pyscf`, `qiskit`, `qiskit-addon-sqd`). + +See [SQD Parameters](#sqd-parameters) below for the full reference, grouped by SQD loop / SBD solver / MPI grid / checkpointing. + +### 3. run_sqd_enlarge_subspace_sbd.py — SQD that grows its own subspace + +Same self-consistent SQD loop as `run_sqd_sbd.py` above (sampling, +configuration recovery, SBD as the solver), but with one addition: between +rounds, it expands the dominant determinant pairs from the just-solved +wavefunction via qiskit-addon-sqd's own `enlarge_batch_from_transitions` +(same-spin single-electron excitations, both alpha and beta), and feeds the +result forward as the next round's `include_configurations`. Concretely, +it calls `diagonalize_fermionic_hamiltonian` with `max_iterations=1` itself, +in its own outer Python loop, rather than delegating the whole +multi-iteration loop to one call the way `run_sqd_sbd.py` does -- that's +what makes injecting a step between rounds possible. The loop stops when +either the expanded set adds nothing new, or the energy and occupancies +both stop moving (`--energy_tol`/`--occupancies_tol`) -- `--max_iterations` +is a safety cap, not the expected stopping mechanism. + +```bash +mpirun -np 8 python -u run_sqd_enlarge_subspace_sbd.py \ + --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ + --counts count_dict_h2o.json \ + --device gpu \ + --adet_comm_size 4 --bdet_comm_size 2 --enlarge_threshold 1e-4 +``` + +**A note on JAX and GPUs.** The expansion step above is JAX. Since XLA reserves 75% of a device +on first use, one rank claims it and the rest fail with `RESOURCE_EXHAUSTED: CUDA_ERROR_OUT_OF_MEMORY`. +On a GPU backend this driver therefore does two things for you, so the command above needs no extra environment: + +- assigns each rank its own device, using `sbd.get_device_id()` so JAX lands + on the same card SBD already selected for that rank; +- sets `XLA_PYTHON_CLIENT_PREALLOCATE=false`, so XLA takes memory on demand + instead of reserving 75% of the card away from SBD. Without this, SBD's + Davidson basis can run out of room at large `--max_dim` and fail inside + `tpb_diag` with `std::bad_alloc` -- a memory error that reads as SBD's + fault but is caused by JAX's reservation. + +`XLA_PYTHON_CLIENT_PREALLOCATE` is left alone if you set it yourself, and +`JAX_PLATFORMS=cpu` still forces the expansion onto CPU. The device assignment +always runs, and stays correct if you pin one GPU per rank yourself: with a +single card visible the count is 1, so every rank resolves to device 0 -- its +own. + +Same bundled 275-bitstring H2O pool as `run_sqd_sbd.py`'s own example above: +plain SQD reaches **≈ -76.236 Ha** and stops there; this driver keeps going +past that fixed pool on its own and converges to **-76.2421767512 Ha**. + +See [SQD Parameters](#sqd-parameters) below for the flags it shares with +`run_sqd_sbd.py` and the ones that differ (`--enlarge_threshold` in place +of `--sqd_carryover_threshold`, and `--max_dim`'s risk profile is sharper +here). + +### 4. run_sqd_sbd.ipynb — Jupyter walkthrough (serial) + +Interactive single-rank companion to `run_sqd_sbd.py`. Same SQD self-consistent +loop on h2o, but inside a Jupyter kernel (`MPI.COMM_WORLD` size 1). Uses the +bundled [`count_dict_h2o.json`](./count_dict_h2o.json) (275 bitstrings → 75,625 +determinants) and reaches ≈ −76.236 Ha in a few seconds on CPU. + +```bash +pytest --nbmake run_sqd_sbd.ipynb # what CI runs; needs the nbtest extra +# or open it in JupyterLab and step through the cells +``` + +## SQD Parameters + +Reference for every flag `run_sqd_sbd.py` accepts, grouped the way `--help` +groups them: SQD loop, SBD solver, MPI grid, checkpointing. + +**How each iteration builds its subspace.** SQD samples bitstrings from a +quantum device, repairs the noisy ones against an orbital-occupancy estimate +(**configuration recovery**), subsamples them into batches, and diagonalizes +each batch. What makes it a *loop* is that two results feed back into the next +iteration. Three sources, concatenated in this priority order inside +qiskit-addon-sqd's `diagonalize_fermionic_hamiltonian` (`fermion.py`): + +``` +strs_a = include_a ++ carryover_strings_a ++ samples_a then dedupe, truncate to max_dim, sort +``` + +1. **`include_a`** — configurations passed as `include_configurations`. Static: + fixed before the loop, present every iteration, never updated. +2. **`carryover_strings_a`** — from the *previous* iteration's wavefunction. Every + determinant whose `|coefficient|` is at least `--sqd_carryover_threshold` + survives, ranked by `|c|^2`. +3. **`samples_a`** — drawn fresh this iteration, sorted by marginal probability. + +The order matters when `max_dim` is set: `include` and `carryover` are kept ahead of +fresh samples, so if those two already fill the cap, this iteration's new samples are +truncated away entirely. + +The samples are not re-used raw counts. Each iteration re-runs configuration +recovery from the *original* bitstrings using the occupancies from the previous +iteration's best batch (`_prepare_ci_strings` in `fermion.py`), then +subsamples. Recovery is **not +cumulative** — it always re-derives from the raw samples, just with a better +occupancy estimate each time. On iteration 1 there are no occupancies yet, so the +raw samples are only filtered by electron count (Hamming-weight postselection). + +So exactly two things flow from iteration N to N+1, and neither is a tolerance: +the **average orbital occupancies** (into recovery, source 3) and the +**wavefunction amplitudes** (into carryover, source 2). + +### SQD loop parameters + +Shared by both `run_sqd_sbd.py` and `run_sqd_enlarge_subspace_sbd.py` except +where noted. **`--max_dim` is the one that most needs attention**: it has no +universally safe default (see below), and in `run_sqd_enlarge_subspace_sbd.py` +leaving it unset is riskier still, since each round's subspace can grow from +the previous one rather than being resampled at a fixed size — the driver +prints an OOM warning when it detects this. + +*Shapes the subspace — changes the numbers you compute:* + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--counts FILE` | Load hardware bitstrings from a JSON file (use this or `--samples`) | none — falls back to `--samples` if omitted | +| `--samples N` | Generate N random bitstrings at the target Hamming weights; plumbing check only, energy not meaningful | `3000` (only used when `--counts` is omitted) | +| `--samples_per_batch` | Dominant control on subspace dimension. With `--symmetrize_spin 1` the alpha and beta string sets are merged, so the subspace is up to `(2N)^2`, not `N^2` | `3000` | +| `--symmetrize_spin` | `1` (default): merge the alpha and beta string pools every iteration, forcing `ci_strs_a == ci_strs_b`. SBD itself supports distinct alpha/beta determinant sets — this is purely a qiskit-addon-sqd loop-layer setting. `0`: sample and carry over alpha and beta independently, allowing them to differ | `1` | +| `--num_batches` | Independent subsamples per iteration; occupancies are averaged across them | `1` (`run_sqd_enlarge_subspace_sbd.py`) / `3` (`run_sqd_sbd.py`) | +| `--sqd_carryover_threshold` | `run_sqd_sbd.py` only. `\|coefficient\|` cutoff for carrying a determinant into the next iteration's sample pool. **Lower it to carry more** | `1e-4` | +| `--enlarge_threshold` | `run_sqd_enlarge_subspace_sbd.py` only — the analogous "carry more" knob for that driver, but structurally different: it gates which *pairs* get expanded into single excitations via `enlarge_batch_from_transitions`, not which determinants survive into resampling. **Lower it to expand more pairs per round** | `1e-4` | +| `--max_dim` | **Critical.** Cap on strings per spin sector, so the subspace cannot exceed `max_dim^2`. The main brake on runaway cost — no fixed value is safe for every system, since the right cap depends on available memory and orbital count. Start from a value known to work at a similar orbital count (e.g. `15000` was used for a 45-orbital system) and adjust down if you see an OOM | unset (no cap) | +| `--include_hf` | Force the single Slater determinant with the lowest `num_elec_a`/`num_elec_b` orbital indices occupied into `include_configurations`, every iteration. Cheap correctness check: that determinant's own diagonal energy is an exact lower bound on what a subspace containing it can do — if forcing it in moves the result, the sampled pool was missing it (and probably its low-excitation neighbors too) | off | + +*Decides when to stop — changes nothing about the subspace:* + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--max_iterations` | Hard cap on loop iterations (not the inner `--sbd_max_it`). In `run_sqd_enlarge_subspace_sbd.py` this is a safety cap only — the loop normally stops earlier, once a round adds no new determinants or both tolerances below are met | `30` (`run_sqd_enlarge_subspace_sbd.py`) / `5` (`run_sqd_sbd.py`) | +| `--energy_tol` | Iteration-to-iteration change in energy | `1e-8` | +| `--occupancies_tol` | Largest change in any single orbital occupancy — an infinity norm, not an average | `1e-5` | + +**Both stopping criteria must hold in the same iteration.** `fermion.py`'s +convergence check combines the energy-change test and the occupancy-change test +with a logical `and`, so the loop only stops once both are satisfied at once — +not whichever one happens first. A run that reaches +`--max_iterations` may be converged in energy while one stubborn orbital's +occupancy is still moving, and loosening only one tolerance will not stop it. +Watch the per-batch energies: while they still disagree, the loop has not +converged regardless of what the total says. + +### SBD solver parameters + +Per diagonalization, not per loop: + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--sbd_eps` | Davidson stop: **norm of the residual vector**, not an energy. Error in the energy goes roughly as `\|R\|^2/gap`, so this already implies far better energy accuracy than `--energy_tol` asks for. Tighten it for a near-degenerate system | `1e-5` | +| `--sbd_max_it` | Cap on Davidson iterations. Reaching it before `--sbd_eps` returns a partially converged vector **with no warning** — watch the `tol=` values SBD prints, and cross-batch agreement | `10` | +| `--sbd_max_nb` | Davidson basis vectors (block size). Peak memory during Davidson grows with how many sub-iterations it actually needs, not just `--max_dim` — a run that succeeds for several iterations at a fixed `dim` can still later need more basis vectors and run out of memory even though the subspace itself did not grow. Lowering this trades some convergence robustness for a lower memory ceiling | `10` | +| `--sbd_method` | 0=Davidson, 1=Davidson+Ham, 2=Lanczos, 3=Lanczos+Ham | `0` | +| `--sbd_use_precalculated_dets` | Thrust only. `1` precomputes a determinant index for **every** (α,β) pair — the whole subspace, on the GPU. `0` uses per-thread storage: slower per matvec, far less memory | `1` | +| `--sbd_max_memory_gb_for_determinants` | Thrust only, and **only consulted when `--sbd_use_precalculated_dets 0`** (`mult_thrust.h:257-273`). Caps the per-thread buffer in GB | `-1` (uncapped) | + +For reference, upstream's own `TPB_SBD` struct defaults are looser still (`max_it=1`, +`eps=1e-4`), and `run_sbd_diag.py` uses `eps=1e-3`. On the h2o counts case, +`eps=1e-5` and `eps=1e-8` give the same energy to ten decimal places. + +### MPI grid parameters + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--adet_comm_size` | Ranks spanning the alpha-determinant dimension | `1` | +| `--bdet_comm_size` | Ranks spanning the beta-determinant dimension | `1` | +| `--task_comm_size` | Ranks spanning task-level parallelism | `1` | + +All ranks diagonalize each batch together, then move to the next batch +sequentially. See [MPI Decomposition](#mpi-decomposition) below for the full 4D +grid (a fourth, *derived* dimension — `helper` — is not set directly). + +### Checkpointing parameters + +| Parameter | What it controls | Default | +|-----------|-----------------|---------| +| `--checkpoint_path` | Write `ci_strs_a`/`ci_strs_b`/`orbital_occupancies`/energy to this path as JSON text (rank 0 only), every `--checkpoint_frequency` iterations. **Must be visible under the same path from every rank** — `--resume_from` has no rank-0-reads-then-broadcasts step, every rank opens this path itself | unset | +| `--checkpoint_frequency` | Write every this many iterations, plus always on the last one regardless of alignment | `1` (every iteration) | +| `--resume_from` | Seed a new run's `include_configurations`/`initial_occupancies` from a previous `--checkpoint_path`'s last recorded iteration | unset | + +What's preserved is which determinants (`ci_strs_a`/`ci_strs_b`, as plain +integers — independent of MPI decomposition, since they carry no rank/grid +information) and the derived per-orbital occupancy averages. **Not** the +wavefunction amplitudes matrix itself (`dim_a × dim_b` floats — at large +`--max_dim` this alone would dwarf everything else; neither consuming parameter +needs it). + +Not a bit-identical continuation: RNG state is fresh in the new process, and +every string from the resumed iteration becomes a **permanent** include for +every iteration of the new run — unlike a true single-process continuation, +where `--sqd_carryover_threshold` would keep pruning low-weight determinants +each iteration. A resumed run is seeded richer than an actual continuation +would have been at that point, not identical to one. + +## MPI Decomposition + +Total MPI ranks must be a **multiple** of +`task_comm_size × adet_comm_size × bdet_comm_size` — not equal to it. SBD splits +the ranks you asked for across those three dimensions and puts whatever remains +into a fourth, "helper" dimension, computed as +`ranks / (task_comm_size × adet_comm_size × bdet_comm_size)`. + +So 8 ranks with `--adet_comm_size 2 --bdet_comm_size 2` is valid: the grid is +`1 × 2 × 2` and the helper dimension absorbs the remaining factor of 2. + +When using more than one rank, specify at least `--adet_comm_size`. Examples: + +| Ranks | Decomposition | Helper | +|-------|---------------|--------| +| 1 | default (all = 1) | 1 | +| 2 | `--adet_comm_size 2` | 1 | +| 4 | `--adet_comm_size 2 --bdet_comm_size 2` | 1 | +| 8 | `--adet_comm_size 2 --bdet_comm_size 2` | 2 | +| 8 | `--adet_comm_size 2 --bdet_comm_size 2 --task_comm_size 2` | 1 | + +## Expected Results + +- **H2O**: ground state energy ≈ **-76.236 Hartree** +- **N2**: ground state energy ≈ **-109.042 Hartree** (with 1e-3 dets) + +## See Also + +- [`../README.md`](../README.md) — backend selection, bundled test data, + performance tips +- [Repository README](../../README.md) — Installation, API reference diff --git a/python/examples/count_dict_h2o.json b/examples/tpb/count_dict_h2o.json similarity index 100% rename from python/examples/count_dict_h2o.json rename to examples/tpb/count_dict_h2o.json diff --git a/python/examples/run_sbd_diag.py b/examples/tpb/run_sbd_diag.py similarity index 100% rename from python/examples/run_sbd_diag.py rename to examples/tpb/run_sbd_diag.py diff --git a/python/examples/run_sqd_enlarge_subspace_sbd.py b/examples/tpb/run_sqd_enlarge_subspace_sbd.py similarity index 100% rename from python/examples/run_sqd_enlarge_subspace_sbd.py rename to examples/tpb/run_sqd_enlarge_subspace_sbd.py diff --git a/python/examples/run_sqd_sbd.ipynb b/examples/tpb/run_sqd_sbd.ipynb similarity index 100% rename from python/examples/run_sqd_sbd.ipynb rename to examples/tpb/run_sqd_sbd.ipynb diff --git a/python/examples/run_sqd_sbd.py b/examples/tpb/run_sqd_sbd.py similarity index 100% rename from python/examples/run_sqd_sbd.py rename to examples/tpb/run_sqd_sbd.py diff --git a/python/__init__.py b/python/__init__.py index c2a683c..f9dee30 100644 --- a/python/__init__.py +++ b/python/__init__.py @@ -465,7 +465,7 @@ def get_device_id(device=None): Uses the communicator rank, i.e. the same ``gpu_id = rank % num_gpus`` convention SBD's own diagonalization applies, documented in - ``python/examples/README.md``. + ``examples/README.md``. """ _ensure_initialized() return get_backend(device).planned_device_id(get_comm()) diff --git a/python/examples/README.md b/python/examples/README.md index 66cbc25..76d65e8 100644 --- a/python/examples/README.md +++ b/python/examples/README.md @@ -1,394 +1,10 @@ -# SQD/SBD Python Examples +# Moved -Examples demonstrating SBD's capabilities for quantum chemistry calculations. +The examples now live at the top level, outside the installed package: -## Overview +- [`examples/tpb/`](../../examples/tpb/README.md) — TPB (tensor-product basis) and + the SQD loops, i.e. everything that used to be in this directory. +- [`examples/README.md`](../../examples/README.md) — backend selection, `--device` + values, bundled test data and performance notes, shared by all examples. -- **Communication:** MPI for distributed computing -- **Backends:** CPU (host OpenMP, `--device cpu`), GPU (NVHPC Thrust, NVIDIA only, `--device gpu`) and GPU (OpenMP target offload, NVIDIA and AMD, `--device gpu-omp`), switchable at runtime via `device` parameter - -Replace `--device gpu` in all the examples below with `--device gpu-omp` if you use AMD GPUs. - -## Examples - -### 1. run_sbd_diag.py — Standalone SBD Diagonalization - -Runs a single TPB diagonalization from an FCIDUMP file and alpha determinant -file. No SQD loop, no Qiskit dependency. - -```bash -# H2O with 2 MPI ranks with CPU -mpirun -np 2 python -u run_sbd_diag.py \ - --device cpu \ - --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ - --adetfile ../../vendor/sbd-upstream/data/h2o/h2o-1em3-alpha.txt \ - --adet_comm_size 2 - -# N2 with GPU -mpirun -np 8 python -u run_sbd_diag.py \ - --device gpu \ - --fcidump ../../vendor/sbd-upstream/data/n2/fcidump.txt \ - --adetfile ../../vendor/sbd-upstream/data/n2/1em3-alpha.txt \ - --adet_comm_size 4 --bdet_comm_size 2 - -# Retrieve the 1-/2-particle RDMs and save them to a file -mpirun -np 2 python -u run_sbd_diag.py --rdm_output /tmp/h2o_rdms.npz -``` - -`--rdm_output` takes the file to save to (`/tmp/h2o_rdms.npz` above) and -writes **one** `.npz` file there holding both `rdm1` and `rdm2` together -(`data = np.load("/tmp/h2o_rdms.npz"); data["rdm1"]`, `data["rdm2"]`) — -unlike upstream SBD's own CLI, which writes two separate files -(`1pRDM.txt`/`2pRDM.txt`). It also prints `trace(rdm1)` and the natural -orbital occupations. Leaving it empty (the default) skips computing RDMs -entirely. - -By default beta determinants are derived from `--adetfile` alone (identical -to it, or a shuffled copy if `--shuffle` is set). `--symmetrize_spin 0` -loads `--adetfile` and `--bdetfile` as independent, genuinely distinct -alpha/beta determinant sets instead; `--bdetfile` is otherwise ignored -(with a warning) since symmetric mode always derives beta from alpha. - -**Key options:** `--device`, `--fcidump`, `--adetfile`, `--bdetfile`, -`--symmetrize_spin`, `--adet_comm_size`, `--bdet_comm_size`, -`--task_comm_size`, `--method`, `--tolerance`, `--iteration`, -`--rdm_output`. (These keep their unprefixed names here: this driver *is* -SBD. The SQD drivers prefix them `--sbd_*`.) Run `python run_sbd_diag.py ---help` for the full list. - -**Requirements:** `sbd`, `mpi4py` - -### 2. run_sqd_sbd.py — SQD Loop with SBD Solver - -Runs the self-consistent SQD workflow (qiskit-addon-sqd) using SBD as the -eigensolver backend. Supports two bitstring input modes: - -- `--counts FILE` — load bitstrings from a count_dict.json -- `--samples N` — generate N random bitstrings at the target Hamming weights - (default). A plumbing check only: random determinants give a random subspace, - so the energy is not meaningful. Use `--counts` for real results. - -```bash -# H2O with the bundled counts file (275 bitstrings -> ~ -76.236 Ha) -mpirun -np 8 python -u run_sqd_sbd.py \ - --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ - --counts count_dict_h2o.json \ - --device gpu \ - --adet_comm_size 4 --bdet_comm_size 2 - -# Custom system with random bitstrings -mpirun -np 8 python -u run_sqd_sbd.py \ - --fcidump /path/to/fci_dump.txt \ - --samples_per_batch 800 --num_batches 3 --max_iterations 10 \ - --device gpu \ - --adet_comm_size 4 --bdet_comm_size 2 -``` - -**count_dict.json format:** A JSON object mapping bitstrings to shot counts, as -produced by a quantum device or simulator. Each bitstring has length `2 × NORB` -and is laid out as **`[beta | alpha]`**: the first `NORB` bits are beta -(spin-down), the last `NORB` are alpha (spin-up), and within each half **orbital 0 -is the rightmost bit**. qiskit-addon-sqd postselects the last `NORB` bits on -`num_elec_a` and the first `NORB` on `num_elec_b`. - -For H2O (NORB=24, 5α+5β) the Hartree–Fock configuration — the five lowest -orbitals doubly occupied — is therefore `"0"*19 + "1"*5` in *both* halves: - -```json -{ - "000000000000000000011111000000000000000000011111": 16, - "010000000010001010000001010000000001000010100100": 12, - "000010001110000000000010001001000110000000000100": 8 -} -``` - -Bitstrings whose halves do not hold exactly `num_elec_a` / `num_elec_b` ones are -dropped by postselection, so a file of uniform-random strings yields nothing -usable — for H2O only `C(24,5)² / 4²⁴ ≈ 6e-6` of them qualify. - -[`count_dict_h2o.json`](./count_dict_h2o.json) in this directory is a ready-made -H2O example: 275 bitstrings taken from the vendored `h2o-1em3-alpha.txt` -determinant list, giving a 275 × 275 = 75,625-determinant subspace at -≈ -76.236 Ha. - -**Key options:** `--fcidump` (required), `--counts`, `--samples`, -`--samples_per_batch`, `--num_batches`, `--max_iterations`, `--device`, -MPI decomposition flags, and the SQD tolerances `--energy_tol` / -`--occupancies_tol` / `--sqd_carryover_threshold`. Inner-solver flags are prefixed -(`--sbd_eps`, `--sbd_max_it`, `--sbd_method`, ...) and have sensible defaults; the -old unprefixed spellings still work. Run `python run_sqd_sbd.py --help` for the -full list, which is grouped by layer. - -**Requirements:** see [Integration with qiskit-addon-sqd](../../README.md#integration-with-qiskit-addon-sqd) in the Python Bindings README (`pyscf`, `qiskit`, `qiskit-addon-sqd`). - -See [SQD Parameters](#sqd-parameters) below for the full reference, grouped by SQD loop / SBD solver / MPI grid / checkpointing. - -### 3. run_sqd_enlarge_subspace_sbd.py — SQD that grows its own subspace - -Same self-consistent SQD loop as `run_sqd_sbd.py` above (sampling, -configuration recovery, SBD as the solver), but with one addition: between -rounds, it expands the dominant determinant pairs from the just-solved -wavefunction via qiskit-addon-sqd's own `enlarge_batch_from_transitions` -(same-spin single-electron excitations, both alpha and beta), and feeds the -result forward as the next round's `include_configurations`. Concretely, -it calls `diagonalize_fermionic_hamiltonian` with `max_iterations=1` itself, -in its own outer Python loop, rather than delegating the whole -multi-iteration loop to one call the way `run_sqd_sbd.py` does -- that's -what makes injecting a step between rounds possible. The loop stops when -either the expanded set adds nothing new, or the energy and occupancies -both stop moving (`--energy_tol`/`--occupancies_tol`) -- `--max_iterations` -is a safety cap, not the expected stopping mechanism. - -```bash -mpirun -np 8 python -u run_sqd_enlarge_subspace_sbd.py \ - --fcidump ../../vendor/sbd-upstream/data/h2o/fcidump.txt \ - --counts count_dict_h2o.json \ - --device gpu \ - --adet_comm_size 4 --bdet_comm_size 2 --enlarge_threshold 1e-4 -``` - -**A note on JAX and GPUs.** The expansion step above is JAX. Since XLA reserves 75% of a device -on first use, one rank claims it and the rest fail with `RESOURCE_EXHAUSTED: CUDA_ERROR_OUT_OF_MEMORY`. -On a GPU backend this driver therefore does two things for you, so the command above needs no extra environment: - -- assigns each rank its own device, using `sbd.get_device_id()` so JAX lands - on the same card SBD already selected for that rank; -- sets `XLA_PYTHON_CLIENT_PREALLOCATE=false`, so XLA takes memory on demand - instead of reserving 75% of the card away from SBD. Without this, SBD's - Davidson basis can run out of room at large `--max_dim` and fail inside - `tpb_diag` with `std::bad_alloc` -- a memory error that reads as SBD's - fault but is caused by JAX's reservation. - -`XLA_PYTHON_CLIENT_PREALLOCATE` is left alone if you set it yourself, and -`JAX_PLATFORMS=cpu` still forces the expansion onto CPU. The device assignment -always runs, and stays correct if you pin one GPU per rank yourself: with a -single card visible the count is 1, so every rank resolves to device 0 -- its -own. - -Same bundled 275-bitstring H2O pool as `run_sqd_sbd.py`'s own example above: -plain SQD reaches **≈ -76.236 Ha** and stops there; this driver keeps going -past that fixed pool on its own and converges to **-76.2421767512 Ha**. - -See [SQD Parameters](#sqd-parameters) below for the flags it shares with -`run_sqd_sbd.py` and the ones that differ (`--enlarge_threshold` in place -of `--sqd_carryover_threshold`, and `--max_dim`'s risk profile is sharper -here). - -### 4. run_sqd_sbd.ipynb — Jupyter walkthrough (serial) - -Interactive single-rank companion to `run_sqd_sbd.py`. Same SQD self-consistent -loop on h2o, but inside a Jupyter kernel (`MPI.COMM_WORLD` size 1). Uses the -bundled [`count_dict_h2o.json`](./count_dict_h2o.json) (275 bitstrings → 75,625 -determinants) and reaches ≈ −76.236 Ha in a few seconds on CPU. - -```bash -pytest --nbmake run_sqd_sbd.ipynb # what CI runs; needs the nbtest extra -# or open it in JupyterLab and step through the cells -``` - -## SQD Parameters - -Reference for every flag `run_sqd_sbd.py` accepts, grouped the way `--help` -groups them: SQD loop, SBD solver, MPI grid, checkpointing. - -**How each iteration builds its subspace.** SQD samples bitstrings from a -quantum device, repairs the noisy ones against an orbital-occupancy estimate -(**configuration recovery**), subsamples them into batches, and diagonalizes -each batch. What makes it a *loop* is that two results feed back into the next -iteration. Three sources, concatenated in this priority order inside -qiskit-addon-sqd's `diagonalize_fermionic_hamiltonian` (`fermion.py`): - -``` -strs_a = include_a ++ carryover_strings_a ++ samples_a then dedupe, truncate to max_dim, sort -``` - -1. **`include_a`** — configurations passed as `include_configurations`. Static: - fixed before the loop, present every iteration, never updated. -2. **`carryover_strings_a`** — from the *previous* iteration's wavefunction. Every - determinant whose `|coefficient|` is at least `--sqd_carryover_threshold` - survives, ranked by `|c|^2`. -3. **`samples_a`** — drawn fresh this iteration, sorted by marginal probability. - -The order matters when `max_dim` is set: `include` and `carryover` are kept ahead of -fresh samples, so if those two already fill the cap, this iteration's new samples are -truncated away entirely. - -The samples are not re-used raw counts. Each iteration re-runs configuration -recovery from the *original* bitstrings using the occupancies from the previous -iteration's best batch (`_prepare_ci_strings` in `fermion.py`), then -subsamples. Recovery is **not -cumulative** — it always re-derives from the raw samples, just with a better -occupancy estimate each time. On iteration 1 there are no occupancies yet, so the -raw samples are only filtered by electron count (Hamming-weight postselection). - -So exactly two things flow from iteration N to N+1, and neither is a tolerance: -the **average orbital occupancies** (into recovery, source 3) and the -**wavefunction amplitudes** (into carryover, source 2). - -### SQD loop parameters - -Shared by both `run_sqd_sbd.py` and `run_sqd_enlarge_subspace_sbd.py` except -where noted. **`--max_dim` is the one that most needs attention**: it has no -universally safe default (see below), and in `run_sqd_enlarge_subspace_sbd.py` -leaving it unset is riskier still, since each round's subspace can grow from -the previous one rather than being resampled at a fixed size — the driver -prints an OOM warning when it detects this. - -*Shapes the subspace — changes the numbers you compute:* - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--counts FILE` | Load hardware bitstrings from a JSON file (use this or `--samples`) | none — falls back to `--samples` if omitted | -| `--samples N` | Generate N random bitstrings at the target Hamming weights; plumbing check only, energy not meaningful | `3000` (only used when `--counts` is omitted) | -| `--samples_per_batch` | Dominant control on subspace dimension. With `--symmetrize_spin 1` the alpha and beta string sets are merged, so the subspace is up to `(2N)^2`, not `N^2` | `3000` | -| `--symmetrize_spin` | `1` (default): merge the alpha and beta string pools every iteration, forcing `ci_strs_a == ci_strs_b`. SBD itself supports distinct alpha/beta determinant sets — this is purely a qiskit-addon-sqd loop-layer setting. `0`: sample and carry over alpha and beta independently, allowing them to differ | `1` | -| `--num_batches` | Independent subsamples per iteration; occupancies are averaged across them | `1` (`run_sqd_enlarge_subspace_sbd.py`) / `3` (`run_sqd_sbd.py`) | -| `--sqd_carryover_threshold` | `run_sqd_sbd.py` only. `\|coefficient\|` cutoff for carrying a determinant into the next iteration's sample pool. **Lower it to carry more** | `1e-4` | -| `--enlarge_threshold` | `run_sqd_enlarge_subspace_sbd.py` only — the analogous "carry more" knob for that driver, but structurally different: it gates which *pairs* get expanded into single excitations via `enlarge_batch_from_transitions`, not which determinants survive into resampling. **Lower it to expand more pairs per round** | `1e-4` | -| `--max_dim` | **Critical.** Cap on strings per spin sector, so the subspace cannot exceed `max_dim^2`. The main brake on runaway cost — no fixed value is safe for every system, since the right cap depends on available memory and orbital count. Start from a value known to work at a similar orbital count (e.g. `15000` was used for a 45-orbital system) and adjust down if you see an OOM | unset (no cap) | -| `--include_hf` | Force the single Slater determinant with the lowest `num_elec_a`/`num_elec_b` orbital indices occupied into `include_configurations`, every iteration. Cheap correctness check: that determinant's own diagonal energy is an exact lower bound on what a subspace containing it can do — if forcing it in moves the result, the sampled pool was missing it (and probably its low-excitation neighbors too) | off | - -*Decides when to stop — changes nothing about the subspace:* - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--max_iterations` | Hard cap on loop iterations (not the inner `--sbd_max_it`). In `run_sqd_enlarge_subspace_sbd.py` this is a safety cap only — the loop normally stops earlier, once a round adds no new determinants or both tolerances below are met | `30` (`run_sqd_enlarge_subspace_sbd.py`) / `5` (`run_sqd_sbd.py`) | -| `--energy_tol` | Iteration-to-iteration change in energy | `1e-8` | -| `--occupancies_tol` | Largest change in any single orbital occupancy — an infinity norm, not an average | `1e-5` | - -**Both stopping criteria must hold in the same iteration.** `fermion.py`'s -convergence check combines the energy-change test and the occupancy-change test -with a logical `and`, so the loop only stops once both are satisfied at once — -not whichever one happens first. A run that reaches -`--max_iterations` may be converged in energy while one stubborn orbital's -occupancy is still moving, and loosening only one tolerance will not stop it. -Watch the per-batch energies: while they still disagree, the loop has not -converged regardless of what the total says. - -### SBD solver parameters - -Per diagonalization, not per loop: - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--sbd_eps` | Davidson stop: **norm of the residual vector**, not an energy. Error in the energy goes roughly as `\|R\|^2/gap`, so this already implies far better energy accuracy than `--energy_tol` asks for. Tighten it for a near-degenerate system | `1e-5` | -| `--sbd_max_it` | Cap on Davidson iterations. Reaching it before `--sbd_eps` returns a partially converged vector **with no warning** — watch the `tol=` values SBD prints, and cross-batch agreement | `10` | -| `--sbd_max_nb` | Davidson basis vectors (block size). Peak memory during Davidson grows with how many sub-iterations it actually needs, not just `--max_dim` — a run that succeeds for several iterations at a fixed `dim` can still later need more basis vectors and run out of memory even though the subspace itself did not grow. Lowering this trades some convergence robustness for a lower memory ceiling | `10` | -| `--sbd_method` | 0=Davidson, 1=Davidson+Ham, 2=Lanczos, 3=Lanczos+Ham | `0` | -| `--sbd_use_precalculated_dets` | Thrust only. `1` precomputes a determinant index for **every** (α,β) pair — the whole subspace, on the GPU. `0` uses per-thread storage: slower per matvec, far less memory | `1` | -| `--sbd_max_memory_gb_for_determinants` | Thrust only, and **only consulted when `--sbd_use_precalculated_dets 0`** (`mult_thrust.h:257-273`). Caps the per-thread buffer in GB | `-1` (uncapped) | - -For reference, upstream's own `TPB_SBD` struct defaults are looser still (`max_it=1`, -`eps=1e-4`), and `run_sbd_diag.py` uses `eps=1e-3`. On the h2o counts case, -`eps=1e-5` and `eps=1e-8` give the same energy to ten decimal places. - -### MPI grid parameters - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--adet_comm_size` | Ranks spanning the alpha-determinant dimension | `1` | -| `--bdet_comm_size` | Ranks spanning the beta-determinant dimension | `1` | -| `--task_comm_size` | Ranks spanning task-level parallelism | `1` | - -All ranks diagonalize each batch together, then move to the next batch -sequentially. See [MPI Decomposition](#mpi-decomposition) below for the full 4D -grid (a fourth, *derived* dimension — `helper` — is not set directly). - -### Checkpointing parameters - -| Parameter | What it controls | Default | -|-----------|-----------------|---------| -| `--checkpoint_path` | Write `ci_strs_a`/`ci_strs_b`/`orbital_occupancies`/energy to this path as JSON text (rank 0 only), every `--checkpoint_frequency` iterations. **Must be visible under the same path from every rank** — `--resume_from` has no rank-0-reads-then-broadcasts step, every rank opens this path itself | unset | -| `--checkpoint_frequency` | Write every this many iterations, plus always on the last one regardless of alignment | `1` (every iteration) | -| `--resume_from` | Seed a new run's `include_configurations`/`initial_occupancies` from a previous `--checkpoint_path`'s last recorded iteration | unset | - -What's preserved is which determinants (`ci_strs_a`/`ci_strs_b`, as plain -integers — independent of MPI decomposition, since they carry no rank/grid -information) and the derived per-orbital occupancy averages. **Not** the -wavefunction amplitudes matrix itself (`dim_a × dim_b` floats — at large -`--max_dim` this alone would dwarf everything else; neither consuming parameter -needs it). - -Not a bit-identical continuation: RNG state is fresh in the new process, and -every string from the resumed iteration becomes a **permanent** include for -every iteration of the new run — unlike a true single-process continuation, -where `--sqd_carryover_threshold` would keep pruning low-weight determinants -each iteration. A resumed run is seeded richer than an actual continuation -would have been at that point, not identical to one. - -## MPI Decomposition - -Total MPI ranks must be a **multiple** of -`task_comm_size × adet_comm_size × bdet_comm_size` — not equal to it. SBD splits -the ranks you asked for across those three dimensions and puts whatever remains -into a fourth, "helper" dimension, computed as -`ranks / (task_comm_size × adet_comm_size × bdet_comm_size)`. - -So 8 ranks with `--adet_comm_size 2 --bdet_comm_size 2` is valid: the grid is -`1 × 2 × 2` and the helper dimension absorbs the remaining factor of 2. - -When using more than one rank, specify at least `--adet_comm_size`. Examples: - -| Ranks | Decomposition | Helper | -|-------|---------------|--------| -| 1 | default (all = 1) | 1 | -| 2 | `--adet_comm_size 2` | 1 | -| 4 | `--adet_comm_size 2 --bdet_comm_size 2` | 1 | -| 8 | `--adet_comm_size 2 --bdet_comm_size 2` | 2 | -| 8 | `--adet_comm_size 2 --bdet_comm_size 2 --task_comm_size 2` | 1 | - -**GDB** (`gdb_diag`) decomposes differently: `t_comm_size × b_comm_size × helper`, -with its own field names rather than TPB's. It is not exercised by these examples -or by the test suite, so its decomposition is unvalidated and is deliberately not -documented further here. - -## Backend Selection - -Every backend the toolchain supported was compiled into this one install, and -each is imported **lazily, on first use** — normally one per process. Select -per-call via `--device`: - -```bash ---device cpu # host OpenMP (default) ---device gpu # NVHPC Thrust (requires NVIDIA GPU + HPC SDK build) ---device gpu-omp # OpenMP target offload, NVIDIA and AMD GPUs ---device auto # GPU if available, else CPU -``` - -`sbd.available_backends()` reports what this install actually has (a static scan -— it does not import anything, so it is safe to call outside `mpirun`), and -`sbd.loaded_backends()` reports what the current process has pulled in. - -Within Python, backends can be selected per call — no re-initialization needed: - -```python -import sbd - -# No init() needed — auto-initializes on first call -result_cpu = sbd.tpb_diag(..., device='cpu') -result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' -# result_omp = sbd.tpb_diag(..., device='gpu-omp') # NOT in the same process as 'cpu' -``` - -## Available Test Data - -**H2O** (`../../vendor/sbd-upstream/data/h2o/`): `h2o-1em3` through `h2o-1em8` alpha determinant files. -**N2** (`../../vendor/sbd-upstream/data/n2/`): `1em3` through `1em7` and `3em4` through `3em7` alpha determinant files. - -Smaller thresholds = more determinants = higher accuracy. - -## Expected Results - -- **H2O**: ground state energy ≈ **-76.236 Hartree** -- **N2**: ground state energy ≈ **-109.042 Hartree** (with 1e-3 dets) - -## Performance Tips - -**CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). - -**GPU:** One MPI rank per GPU, `OMP_NUM_THREADS=1`. Each rank auto-assigned: `gpu_id = rank % num_gpus`. Use method 0 (matrix-free Davidson) for best GPU performance. - -## See Also - -- [Python Bindings README](../../README.md) — Installation, API reference -- [Upstream SBD library](https://github.com/r-ccs-cms/sbd) — C++ library overview +This file is a signpost for older links and can be removed once they have aged out. diff --git a/tox.ini b/tox.ini index dd3ecb9..0f98e64 100644 --- a/tox.ini +++ b/tox.ini @@ -69,7 +69,7 @@ extras = nbtest notebook-dependencies commands = - pytest --nbmake --nbmake-timeout=3000 {posargs} python/examples/ + pytest --nbmake --nbmake-timeout=3000 {posargs} examples/ [testenv:slow] # The reference cases grow by roughly an order of magnitude in determinant count per From 8306ed20b4ab4f15ef73c86f973f51ded2e83063 Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 13:27:11 -0400 Subject: [PATCH 02/10] docs: split the install matrix out to INSTALL.md Installation was 190 of the README's 424 lines -- 45% of the landing page spent on prerequisites, every environment variable the build reads, host-MPI builds and verification, before a reader learned what the package does. It moves to INSTALL.md, taking the README to 264 lines. The README keeps a quickstart rather than deferring everything, because it is also the PyPI long_description: a conda env, pip install, and the available_backends() check that tells you which backends the build produced, plus the GPU-aware MPI requirement, which is the one prerequisite easy to miss and expensive to diagnose. Everything else -- per-platform prerequisites, installing from a checkout with the submodule, the full environment-variable matrix, narrowing which backends get built, building against an existing host MPI -- is one link away. Extracted subsections are promoted to top-level headings, the one intra-README reference ("see Backend Architecture below") now names the README explicitly, and INSTALL.md is added to MANIFEST.in so it reaches the sdist. Verified: every relative link in both files resolves, the three referenced anchors exist, and the sdist contains INSTALL.md alongside the READMEs. Co-Authored-By: Claude Opus 5 (1M context) --- INSTALL.md | 202 ++++++++++++++++++++++++++++++++++++++++++++++++++++ MANIFEST.in | 1 + README.md | 188 ++++-------------------------------------------- 3 files changed, 217 insertions(+), 174 deletions(-) create mode 100644 INSTALL.md diff --git a/INSTALL.md b/INSTALL.md new file mode 100644 index 0000000..08e53a3 --- /dev/null +++ b/INSTALL.md @@ -0,0 +1,202 @@ +# Installing sbd-eigensolver + +Every install compiles the extension modules on the target machine — no wheels are +published — so which backends you end up with depends on the toolchain the build +finds. The quickstart in the [README](README.md#installation) covers the common CPU +case; this page is the full matrix: prerequisites, the environment variables the build +reads, building against a host MPI, and how to verify what you got. + +## Prerequisites + +**Required:** Python 3.10+, MPI (OpenMPI/MPICH), BLAS (OpenBLAS/MKL), pybind11, mpi4py, numpy, compiler with OpenMP. + +**On macOS:** Apple clang ships without OpenMP, so add `llvm-openmp` to the conda +environment. Homebrew's `libomp` is used as a fallback if the env has none. Pin the +compiler with `CC`/`CXX` as well — a bare `clang++` is resolved through `PATH`, so a +Homebrew LLVM silently wins over both Apple clang and a conda toolchain. The build +prints which compiler and which libomp it chose. + +**For the GPU backends** — optional; without them you get a CPU-only install: + +***On NVIDIA*** (both the Thrust and the OpenMP-offload backend): + +- *to build:* [NVIDIA HPC SDK](https://developer.nvidia.com/hpc-sdk) (`nvc++`). + `SBD_GPU_ARCH` is **optional**: unset, nvc++ targets the GPU of the machine + the toolchain was installed on; set, it is honored exactly and may name + several generations at once (`cc80,cc90,cc100`) — see + [Environment Variables](#environment-variables). + +***On AMD*** (the OpenMP-offload backend only): + +- *to build:* ROCm LLVM toolchain (`amdclang++`). + `SBD_GPU_ARCH` is optional here on a GPU system and is **detected** with + ROCm's `amdgpu-arch` (e.g. `gfx90a` on MI250X, `gfx942` on MI300X). However, on a GPU-less build host, you must + set it. see [Environment Variables](#environment-variables). + +Either install path compiles the C++ extension on the target machine; +no pre-built wheels are published. The resulting binary depends on the +local MPI and BLAS, so both must be installed first. + +## Install from PyPI + +**A self-contained conda environment** is the quickest way to get those +dependencies in place for the CPU backend: +```bash +conda create -y -n sbd -c conda-forge \ + python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip mpi4py +# ...plus llvm-openmp on macOS +``` + +```bash +conda activate sbd +``` + +Now install the sbd-eigensolver-python package +``` +pip install sbd-eigensolver +``` + +The published source distribution (sdist) bundles the sbd header files, so this +needs no git checkout and no submodule step. + +## Install from git checkout + +Here the headers come from upstream +[r-ccs-cms/sbd](https://github.com/r-ccs-cms/sbd) via a git submodule at +`vendor/sbd-upstream/`, **pinned at a specific upstream commit**. Run `git submodule status` to see the pinned SHA. + +When installing from a git checkout, it is important to make sure the +sbd submodule is cloned, too: + +```bash +git clone --recurse-submodules https://github.com/Qiskit/sbd-eigensolver-python.git +# or, if you cloned without --recurse-submodules: +git submodule update --init --recursive +``` + +If you need a newer upstream revision (for a recently-landed GPU fix +etc.), advance the local submodule and rebuild: + +```bash +git submodule update --remote vendor/sbd-upstream +``` + +After the upstream sbd code is cloned, run: +``` +pip install -e . --no-build-isolation --force-reinstall --no-deps +``` + +## Environment Variables + +```bash +# --- NVIDIA GPU backends (Thrust and OpenMP target-offload): point at NVHPC. +# Only needed if nvc++ is not already on PATH. Adjust the path. +export NVHPC_HOME=/opt/nvidia/hpc_sdk/Linux_x86_64/2025/compilers + +# --- AMD GPU backend (OpenMP target-offload): point at ROCm LLVM toolchain +# amdclang++ is found on $ROCM_HOME/bin +export ROCM_HOME=/opt/rocm + +# --- OPTIONAL. Only for a host with BOTH GPU toolchains installed, where the +# auto-answer would be an accident of probe order: nvidia | amd | none +export SBD_GPU_VENDOR=amd + +# --- OPTIONAL GPU architecture, spelled per vendor. +# NVIDIA: unset, nvc++ targets the GPU of the machine this toolchain was +# installed on. Set it to pin the target, including SEVERAL at once: +# A100: cc80 H100: cc90 GB200 / B200: cc100 +# ccall-major one target per major generation +# AMD: unset, the arch is DETECTED with ROCm's `amdgpu-arch`. Set it to pin, +# or when building on a host with no GPU (where detection cannot work and +# the build stops asking for it): +# MI250X: gfx90a MI300X: gfx942 (several: gfx90a,gfx942) +export SBD_GPU_ARCH=cc80,cc90,cc100 # NVIDIA +export SBD_GPU_ARCH=gfx90a # AMD + +# --- optional overrides; each has a working default --- +# Which backends to build: defaults to CPU always, plus every GPU backend the +# detected toolchain supports -- Thrust AND OpenMP-offload under nvc++, +# OpenMP-offload only under amdclang++ (there is no rocThrust path). +# Set it only to narrow that: +# cpu CPU only -- skip GPU even if a GPU compiler is present +# gpu Thrust GPU only, no CPU -- NVIDIA only, errors on AMD +# gpu_omp_offload OpenMP target-offload GPU only (either vendor) +export SBD_BUILD_BACKEND=cpu + +# MPI: defaults to whatever mpi4py is linked against. Set this only +# for layouts that cannot be inferred. +# NOTE: Every GPU backend hands MPI device pointers, so use a GPU-aware MPI: +# CUDA-aware on NVIDIA, ROCm-aware on AMD. +export MPI_HOME=/path/to/mpi + +# BLAS: defaults to whatever the linker finds, including a +# conda-installed OpenBLAS in $CONDA_PREFIX/lib. Set these to select +# a specific build (e.g. an arch-tuned OpenBLAS) +export BLAS_LIB_PATH=/path/to/blas/lib +export BLAS_LIBS=openblas # or mkl_rt + +# these are read while COMPILING, so set them first, then install +pip install sbd-eigensolver # from PyPI +pip install -e . --no-build-isolation --no-deps # from a git checkout +``` + +## Build Using the Host MPI +```bash +# Create a conda env +conda create -y -n sbd -c conda-forge \ + python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip +# ...plus llvm-openmp on macOS + +conda activate sbd # always activate first + +# Install mpi4py against the host MPI +# For any GPU backend the host MPI must be GPU-aware: +# CUDA-aware on NVIDIA, ROCm-aware on AMD. +export MPI_HOME=/path/to/mpi +MPICC=$MPI_HOME/bin/mpicc python -m pip install --no-binary=mpi4py --no-cache-dir mpi4py +# NOTE: If no host MPI is available, let conda pick a compatible one with the +# command below. A default conda-forge MPI is not GPU-aware, which is fine for the +# CPU backend; for the GPU backends either install mpi4py against a GPU-aware MPI +# as above, or see the README's Backend Architecture section for the two build-time options that +# let the Thrust backend run without one. +# conda install -y -c conda-forge mpi4py + +# confirm which MPI mpi4py uses -- setup.py builds against exactly this +python -c "from mpi4py import MPI; print(MPI.Get_library_version())" + +# only for the SQD examples (examples/tpb/run_sqd_sbd.py and .ipynb) +pip install "qiskit-addon-sqd>=0.13.1" + +# install sbd-eigensolver +pip install sbd-eigensolver + +``` + +## Verify + +```bash +python -c "import sbd; print(sbd.available_backends())" +# CPU only: ['cpu'] +# NVIDIA, default build: ['cpu', 'gpu', 'gpu-omp'] +# AMD, default build: ['cpu', 'gpu-omp'] +# OMP-offload-only install: ['gpu-omp'] +``` + +On a GPU build, confirm which vendor and architecture the offload backend +targets — one `'gpu-omp'` device serves both vendors, so the name alone does not +say: + +```bash +python -c "import sbd; print(sbd.get_backend('gpu-omp').__sbd_offload_target__)" +# amdgcn-amd-amdhsa:gfx90a AMD MI250X +# nvptx64-nvidia-cuda:cc90 NVIDIA H100 +``` + +The Thrust backend is stamped too (`cuda:cc90`); the CPU backend reports `None`. + +## See Also + +- [README](README.md) — what the package is, the API, and the SQD integration +- [`examples/README.md`](examples/README.md) — backend selection at runtime, bundled + test data and performance notes +- [Troubleshooting](README.md#troubleshooting) — build and runtime symptoms diff --git a/MANIFEST.in b/MANIFEST.in index 4d49a21..6bb3f07 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -18,3 +18,4 @@ include vendor/sbd-upstream/LICENSE.txt # Example scripts and notebook, referenced by the README. recursive-include examples *.py *.ipynb *.md *.json include python/examples/README.md +include INSTALL.md diff --git a/README.md b/README.md index 6d41dac..52665ff 100644 --- a/README.md +++ b/README.md @@ -23,193 +23,33 @@ In addition to TPB, this package also contains experimental support for SBD's ** ## Installation -### Prerequisites - -**Required:** Python 3.10+, MPI (OpenMPI/MPICH), BLAS (OpenBLAS/MKL), pybind11, mpi4py, numpy, compiler with OpenMP. - -**On macOS:** Apple clang ships without OpenMP, so add `llvm-openmp` to the conda -environment. Homebrew's `libomp` is used as a fallback if the env has none. Pin the -compiler with `CC`/`CXX` as well — a bare `clang++` is resolved through `PATH`, so a -Homebrew LLVM silently wins over both Apple clang and a conda toolchain. The build -prints which compiler and which libomp it chose. - -**For the GPU backends** — optional; without them you get a CPU-only install: - -***On NVIDIA*** (both the Thrust and the OpenMP-offload backend): - -- *to build:* [NVIDIA HPC SDK](https://developer.nvidia.com/hpc-sdk) (`nvc++`). - `SBD_GPU_ARCH` is **optional**: unset, nvc++ targets the GPU of the machine - the toolchain was installed on; set, it is honored exactly and may name - several generations at once (`cc80,cc90,cc100`) — see - [Environment Variables](#environment-variables). - -***On AMD*** (the OpenMP-offload backend only): - -- *to build:* ROCm LLVM toolchain (`amdclang++`). - `SBD_GPU_ARCH` is optional here on a GPU system and is **detected** with - ROCm's `amdgpu-arch` (e.g. `gfx90a` on MI250X, `gfx942` on MI300X). However, on a GPU-less build host, you must - set it. see [Environment Variables](#environment-variables). - -Either install path compiles the C++ extension on the target machine; -no pre-built wheels are published. The resulting binary depends on the -local MPI and BLAS, so both must be installed first. - -### Install from PyPI - -**A self-contained conda environment** is the quickest way to get those -dependencies in place for the CPU backend: -```bash -conda create -y -n sbd -c conda-forge \ - python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip mpi4py -# ...plus llvm-openmp on macOS -``` +The extension modules are compiled on your machine — no wheels are published — so the +backends you get depend on the toolchain the build finds. For the common CPU case: ```bash +# an environment with a compiler, MPI and BLAS (plus llvm-openmp on macOS) +conda create -n sbd -c conda-forge python=3.12 mpi4py openblas conda activate sbd -``` - -Now install the sbd-eigensolver-python package -``` -pip install sbd-eigensolver -``` - -The published source distribution (sdist) bundles the sbd header files, so this -needs no git checkout and no submodule step. - -### Install from git checkout - -Here the headers come from upstream -[r-ccs-cms/sbd](https://github.com/r-ccs-cms/sbd) via a git submodule at -`vendor/sbd-upstream/`, **pinned at a specific upstream commit**. Run `git submodule status` to see the pinned SHA. -When installing from a git checkout, it is important to make sure the -sbd submodule is cloned, too: - -```bash -git clone --recurse-submodules https://github.com/Qiskit/sbd-eigensolver-python.git -# or, if you cloned without --recurse-submodules: -git submodule update --init --recursive -``` - -If you need a newer upstream revision (for a recently-landed GPU fix -etc.), advance the local submodule and rebuild: - -```bash -git submodule update --remote vendor/sbd-upstream -``` - -After the upstream sbd code is cloned, run: -``` -pip install -e . --no-build-isolation --force-reinstall --no-deps -``` - -### Environment Variables - -```bash -# --- NVIDIA GPU backends (Thrust and OpenMP target-offload): point at NVHPC. -# Only needed if nvc++ is not already on PATH. Adjust the path. -export NVHPC_HOME=/opt/nvidia/hpc_sdk/Linux_x86_64/2025/compilers - -# --- AMD GPU backend (OpenMP target-offload): point at ROCm LLVM toolchain -# amdclang++ is found on $ROCM_HOME/bin -export ROCM_HOME=/opt/rocm - -# --- OPTIONAL. Only for a host with BOTH GPU toolchains installed, where the -# auto-answer would be an accident of probe order: nvidia | amd | none -export SBD_GPU_VENDOR=amd - -# --- OPTIONAL GPU architecture, spelled per vendor. -# NVIDIA: unset, nvc++ targets the GPU of the machine this toolchain was -# installed on. Set it to pin the target, including SEVERAL at once: -# A100: cc80 H100: cc90 GB200 / B200: cc100 -# ccall-major one target per major generation -# AMD: unset, the arch is DETECTED with ROCm's `amdgpu-arch`. Set it to pin, -# or when building on a host with no GPU (where detection cannot work and -# the build stops asking for it): -# MI250X: gfx90a MI300X: gfx942 (several: gfx90a,gfx942) -export SBD_GPU_ARCH=cc80,cc90,cc100 # NVIDIA -export SBD_GPU_ARCH=gfx90a # AMD - -# --- optional overrides; each has a working default --- -# Which backends to build: defaults to CPU always, plus every GPU backend the -# detected toolchain supports -- Thrust AND OpenMP-offload under nvc++, -# OpenMP-offload only under amdclang++ (there is no rocThrust path). -# Set it only to narrow that: -# cpu CPU only -- skip GPU even if a GPU compiler is present -# gpu Thrust GPU only, no CPU -- NVIDIA only, errors on AMD -# gpu_omp_offload OpenMP target-offload GPU only (either vendor) -export SBD_BUILD_BACKEND=cpu - -# MPI: defaults to whatever mpi4py is linked against. Set this only -# for layouts that cannot be inferred. -# NOTE: Every GPU backend hands MPI device pointers, so use a GPU-aware MPI: -# CUDA-aware on NVIDIA, ROCm-aware on AMD. -export MPI_HOME=/path/to/mpi - -# BLAS: defaults to whatever the linker finds, including a -# conda-installed OpenBLAS in $CONDA_PREFIX/lib. Set these to select -# a specific build (e.g. an arch-tuned OpenBLAS) -export BLAS_LIB_PATH=/path/to/blas/lib -export BLAS_LIBS=openblas # or mkl_rt - -# these are read while COMPILING, so set them first, then install -pip install sbd-eigensolver # from PyPI -pip install -e . --no-build-isolation --no-deps # from a git checkout -``` - -### Build Using the Host MPI -```bash -# Create a conda env -conda create -y -n sbd -c conda-forge \ - python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip -# ...plus llvm-openmp on macOS - -conda activate sbd # always activate first - -# Install mpi4py against the host MPI -# For any GPU backend the host MPI must be GPU-aware: -# CUDA-aware on NVIDIA, ROCm-aware on AMD. -export MPI_HOME=/path/to/mpi -MPICC=$MPI_HOME/bin/mpicc python -m pip install --no-binary=mpi4py --no-cache-dir mpi4py -# NOTE: If no host MPI is available, let conda pick a compatible one with the -# command below. A default conda-forge MPI is not GPU-aware, which is fine for the -# CPU backend; for the GPU backends either install mpi4py against a GPU-aware MPI -# as above, or see "Backend Architecture" below for the two build-time options that -# let the Thrust backend run without one. -# conda install -y -c conda-forge mpi4py - -# confirm which MPI mpi4py uses -- setup.py builds against exactly this -python -c "from mpi4py import MPI; print(MPI.Get_library_version())" - -# only for the SQD examples (examples/tpb/run_sqd_sbd.py and .ipynb) -pip install "qiskit-addon-sqd>=0.13.1" - -# install sbd-eigensolver pip install sbd-eigensolver - ``` -### Verify +Then check what was built: ```bash python -c "import sbd; print(sbd.available_backends())" -# CPU only: ['cpu'] -# NVIDIA, default build: ['cpu', 'gpu', 'gpu-omp'] -# AMD, default build: ['cpu', 'gpu-omp'] -# OMP-offload-only install: ['gpu-omp'] +# CPU only: ['cpu'] +# NVIDIA, default build: ['cpu', 'gpu', 'gpu-omp'] +# AMD, default build: ['cpu', 'gpu-omp'] ``` -On a GPU build, confirm which vendor and architecture the offload backend -targets — one `'gpu-omp'` device serves both vendors, so the name alone does not -say: - -```bash -python -c "import sbd; print(sbd.get_backend('gpu-omp').__sbd_offload_target__)" -# amdgcn-amd-amdhsa:gfx90a AMD MI250X -# nvptx64-nvidia-cuda:cc90 NVIDIA H100 -``` +**See [INSTALL.md](INSTALL.md)** for the rest: prerequisites per platform, installing +from a git checkout with the submodule, every environment variable the build reads +(GPU toolchains, architectures, MPI and BLAS selection, narrowing which backends get +built), building against an existing host MPI, and fuller verification. -The Thrust backend is stamped too (`cuda:cc90`); the CPU backend reports `None`. +GPU backends need a **GPU-aware MPI** — CUDA-aware on NVIDIA, ROCm-aware on AMD — +because every one of them hands MPI device pointers. ## Examples From 75371fd1bc931115a8bf2726283b0c695ff65a8b Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 14:48:10 -0400 Subject: [PATCH 03/10] README: drop an over-general MPI claim, and say whose limit each GDB constraint is Two corrections. The install quickstart asserted that GPU backends "need a GPU-aware MPI ... because every one of them hands MPI device pointers". That is wrong twice over: it is not every backend -- the OpenMP-offload path works with an MPI that cannot address device memory, since ROCm device allocations are host-addressable -- and it is not a hard requirement even for Thrust, which has two documented build-time escape hatches (SBD_NON_CUDA_AWARE_MPI, SBD_THRUST_SAFE_MPI_ALLREDUCE). The Backend Architecture section already states it accurately as an assumption of the Thrust build with those hatches, and INSTALL.md carries the detail, so the quickstart line is removed rather than reworded. The GDB config table said b_comm_size "must be 1" and left t_comm_size unexplained, which invites the reader to assume both are SBD limitations. They are not the same kind of thing. b_comm_size == 1 is OUR calling convention: upstream's in-memory gdb::diag expects each rank to pass its own shard -- that is what its file-based path feeds it after distributing files over b_comm -- while this wrapper passes the whole list from every rank, and upstream's own run.sh for the GDB app uses --b_comm_size 2. t_comm_size == 1 is then UPSTREAM's algorithm: the matvec rotates the ket around b_comm as a ring, so there are exactly b_comm_size stations and one task is one station, hence t_comm_size cannot exceed b_comm_size. Also notes that the derived helper dimension still absorbs the remaining ranks. Co-Authored-By: Claude Opus 5 (1M context) --- README.md | 26 +++++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 52665ff..71fba46 100644 --- a/README.md +++ b/README.md @@ -48,9 +48,6 @@ from a git checkout with the submodule, every environment variable the build rea (GPU toolchains, architectures, MPI and BLAS selection, narrowing which backends get built), building against an existing host MPI, and fuller verification. -GPU backends need a **GPU-aware MPI** — CUDA-aware on NVIDIA, ROCm-aware on AMD — -because every one of them hands MPI device pointers. - ## Examples Located in [`examples/`](examples/README.md), organized by basis type since the @@ -201,13 +198,32 @@ and replaces the determinant communicators with a single basis communicator: | Attribute | Default | Description | |-----------|---------|-------------| -| `b_comm_size` | 1 | Basis communicator size (must be 1 for `gdb_diag`) | -| `t_comm_size` | 1 | Task communicator size | +| `b_comm_size` | 1 | Basis communicator size — must be 1 for `gdb_diag`, see below | +| `t_comm_size` | 1 | Task communicator size — must be 1 while `b_comm_size` is, see below | | `seed` | 1729 | Seed for the initial vector | | `heatbath_cutoff` | 1e-4 | Heatbath expansion cutoff | | `heatbath_truncation` | 0.0 | Weight truncation applied before heatbath expansion | | `heatbath_batch_size` | 200000000 | Heatbath expansion batch size | +**Why both must be 1**, since the two constraints have different owners: + +`b_comm_size == 1` is a limitation of *this wrapper*, not of SBD. Upstream's in-memory +`gdb::diag` expects each rank to pass **its own shard** of the determinant list — that +is what upstream's file-based entry point hands it, after distributing determinant +files across `b_comm`. This wrapper passes the whole list from every rank, which is +only consistent with a single basis block, so it rejects anything else rather than +have each rank diagonalize the full subspace while believing it held a shard. Upstream +itself runs with a split basis: its own `run.sh` for the GDB app passes +`--b_comm_size 2`. + +`t_comm_size == 1` then follows from *upstream's* algorithm rather than from us. GDB's +matrix-vector product rotates the ket around `b_comm` as a ring, so there are exactly +`b_comm_size` ring stations and one "task" is one station — meaning `t_comm_size` +cannot exceed `b_comm_size`. With the basis in a single block there is a single task. + +Ranks are not wasted in the meantime: the derived helper dimension, +`ranks / (t_comm_size × b_comm_size)`, absorbs them and does not change the energy. + ### Diagonalization ```python From 239001043de0c8c8a3e36585f6f81aaddf5af661 Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 14:58:30 -0400 Subject: [PATCH 04/10] README: use the documented conda command in the quickstart, not an invented one When splitting install out to INSTALL.md I wrote a shortened environment command from scratch instead of reusing the one the project documents. It diverged four ways: python=3.12 instead of the pinned 3.13.12, no -y, no pybind11/numpy/setuptools/wheel/ pip/pyscf, and -- the one that actually matters -- it dropped the "plus llvm-openmp on macOS" note. setup.py's Darwin path looks for omp.h in the conda env, so a macOS reader following the quickstart would have hit exactly the OpenMP problem issue #27 is about. The build-tool packages are arguably redundant under pip's build isolation, since pyproject's [build-system] requires supplies them, but that is not a judgement to make silently in a quickstart -- and two different install commands in one repository is a maintenance trap regardless. Now verbatim from INSTALL.md. Co-Authored-By: Claude Opus 5 (1M context) --- README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 71fba46..5c16b31 100644 --- a/README.md +++ b/README.md @@ -27,8 +27,9 @@ The extension modules are compiled on your machine — no wheels are published backends you get depend on the toolchain the build finds. For the common CPU case: ```bash -# an environment with a compiler, MPI and BLAS (plus llvm-openmp on macOS) -conda create -n sbd -c conda-forge python=3.12 mpi4py openblas +conda create -y -n sbd -c conda-forge \ + python=3.13.12 pybind11 numpy setuptools wheel openblas pyscf pip mpi4py +# ...plus llvm-openmp on macOS conda activate sbd pip install sbd-eigensolver From 95d89532324094aaa3eddb9926769e74334ed195 Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 15:03:24 -0400 Subject: [PATCH 05/10] README: audit the troubleshooting entries against the code Checked all five against what the build and bindings actually do now. Three hold unchanged, one was stale, and two symptoms we know about were missing. Still accurate, verified: the conda CC/CXX entry (setup.py uses os.environ.setdefault, so a caller-set compiler still suppresses nvc++/amdclang++); the GPU-not-building entry; and the Bus error / SIGSEGV entry, including both of its misleading cases. NOT resolved, contrary to how it might read: the "all ranks land on GPU 0" entry. The omp_get_num_devices() fallback to counting the vendor visibility variable is still there, and so is omp_set_default_device(mpi_rank % n_dev). Added the caveat that the index is the GLOBAL rank, so the assignment is only even when the launcher places ranks on nodes in contiguous blocks; round-robin placement leaves each node on a strided subset of its GPUs. Stale: "MPI errors: verify MPI_HOME, check MPI.Get_version()" predates the build learning to diagnose this. setup.py now stops with "MPI_HOME=... is not the MPI that mpi4py is linked against", so the entry now names that message, explains why it is fatal rather than a warning, and gives the two resolutions. Added, both real and previously undocumented here: the macOS OMP Error #15 duplicate libomp abort, which kills the first diagonalization while letting the import succeed (issue #27); and multi-rank GPU GDB failing with "GDB Thrust mult does not support h_comm_size > 1", which is unavoidable in this release because b_comm_size is pinned to 1 and every added rank therefore lands in the helper dimension. Co-Authored-By: Claude Opus 5 (1M context) --- README.md | 30 +++++++++++++++++++++++++++--- 1 file changed, 27 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 5c16b31..e9e4131 100644 --- a/README.md +++ b/README.md @@ -272,10 +272,34 @@ picked (`Found amdclang++ in PATH: …` / `Found NVIDIA HPC SDK at: …`) and, f the offload backend, the resolved architecture; on a host with both toolchains force the choice with `SBD_GPU_VENDOR=amd|nvidia`. -**MPI errors:** Verify `MPI_HOME`, check `python -c "from mpi4py import MPI; print(MPI.Get_version())"`. - -**OMP-offload runs all land on GPU 0 in multi-GPU jobs:** symptom — every MPI rank shows large memory only on GPU 0 in `nvidia-smi` (or `rocm-smi`). The bindings call `omp_set_default_device(mpi_rank % n_dev)`, but `omp_get_num_devices()` can return 0 in some dlopen scenarios. The bindings fall back to counting the entries in the vendor's device-visibility variable — `CUDA_VISIBLE_DEVICES` on NVIDIA, `ROCR_VISIBLE_DEVICES` or `HIP_VISIBLE_DEVICES` on AMD — so make sure the relevant one is exported and lists all your GPUs (e.g. `0,1,2,3`). Slurm/`srun --gres=gpu:N` and OpenMPI's default binding policy already do this; if you've custom-restricted it to a single GPU per rank, set it manually before launch. +**`MPI_HOME=... is not the MPI that mpi4py is linked against`:** the build stops here +deliberately rather than producing extensions that link one MPI while `mpi4py` loads +another — a mismatch that surfaces later as undefined symbols or a hang inside the first +collective. `MPI_HOME` is only for layouts the build cannot infer; unset it to use +`mpi4py`'s own MPI, or reinstall `mpi4py` against the MPI you want +(`pip install --no-binary :all: mpi4py`). To see which MPI that is: +`python -c "from mpi4py import MPI; print(MPI.Get_library_version())"`. + +**`OMP: Error #15: Initializing libomp.dylib, but found libomp.dylib already +initialized` on macOS:** two copies of the same LLVM OpenMP runtime in one process. It +aborts at the first parallel region, so the import succeeds and the first +diagonalization dies. Usually it means the environment provides `libomp` twice — for +example Homebrew `llvm` *and* Homebrew `libomp`, or a Homebrew copy alongside the conda +env's. Build against one only; the conda env's is the one loaded at import time, so +prefer it. Tracked as +[issue #27](https://github.com/Qiskit/sbd-eigensolver-python/issues/27). + +**OMP-offload runs all land on GPU 0 in multi-GPU jobs:** symptom — every MPI rank shows large memory only on GPU 0 in `nvidia-smi` (or `rocm-smi`). The bindings call `omp_set_default_device(mpi_rank % n_dev)`, but `omp_get_num_devices()` can return 0 in some dlopen scenarios. The bindings fall back to counting the entries in the vendor's device-visibility variable — `CUDA_VISIBLE_DEVICES` on NVIDIA, `ROCR_VISIBLE_DEVICES` or `HIP_VISIBLE_DEVICES` on AMD — so make sure the relevant one is exported and lists all your GPUs (e.g. `0,1,2,3`). Slurm/`srun --gres=gpu:N` and OpenMPI's default binding policy already do this; if you've custom-restricted it to a single GPU per rank, set it manually before launch. Note the index is the **global** MPI rank, not a node-local one, so the assignment is even only when the launcher places ranks on nodes in contiguous blocks — round-robin placement leaves each node using a strided subset of its GPUs. **Ranks die with `Bus error` or `SIGSEGV` inside the MPI's own copy path** (`MPIR_Localcopy`, `ucp_worker_progress`, ...) **on a GPU backend:** the MPI is not GPU-aware and was handed a device pointer. Rebuild UCX `--with-cuda` / `--with-rocm`, and confirm with `ucx_info -d | grep -i 'Transport: cuda'` (or `rocm`). Two things mislead here. A partly GPU-aware stack fails in only one place: an MPICH with GPU support *disabled* over a CUDA-aware UCX ran OMP-offload fine and crashed only in Thrust, because the inter-rank path went through UCX while the local-copy path did not. And on AMD a non-ROCm-aware MPI does not crash at all — ROCm maps device memory into the process address space, so the host copy succeeds and merely stages everything through the host aperture (measured on MI250X, XNACK off, 8 ranks) — so a working AMD run is not evidence that the MPI is ROCm-aware. +**`GDB Thrust mult does not support h_comm_size > 1` from `gdb_diag` on more than one +rank:** GDB's Thrust kernels never implemented the helper dimension, and the helper +dimension is `ranks / (t_comm_size × b_comm_size)`. Since `gdb_diag` requires +`b_comm_size == 1` (see [Configuration](#configuration)), which forces `t_comm_size` to +1, every rank you add lands in the helper dimension — so GPU GDB is limited to a single +rank in this release. Run GDB on one GPU, or on the CPU backend, where the helper +dimension is unconstrained. TPB is unaffected and shards over `adet_comm_size` / +`bdet_comm_size` as usual. + **Repository:** https://github.com/Qiskit/sbd-eigensolver-python From 2965adfec89a2db17c2bac661a30ddb74a61c77b Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 15:17:38 -0400 Subject: [PATCH 06/10] examples/README: explain the cpu/gpu-omp conflict, and stop advising one thread on GPU Two things a reader could not act on. "# NOT in the same process as 'cpu'" said don't without saying what happens. Both backends link the same OpenMP runtime and _core_cpu is built without offload support, so whichever loads first initializes it host-only; if that is the CPU backend, the OMP-offload backend cannot acquire a device and silently runs its target regions on the host -- right answers, exit 0, idle GPU, nothing to catch. Now stated, with loaded_backends() as the way to check, and with the note that 'cpu' and 'gpu' (Thrust) do coexist because Thrust does not route device work through OpenMP. "GPU: One MPI rank per GPU, OMP_NUM_THREADS=1" is wrong advice. One rank per GPU is about device ownership; it says nothing about threads, and a GPU build still does real host work. Helper construction is host-threaded in both solvers (tpb/helper.h has 15 omp regions, gdb/helper.h one), and for GDB the heatbath expansion and carryover selection have no device implementation at all -- gdb/expansion.h and gdb/carryover.h contain no thrust:: code and are included outside any SBD_THRUST guard. One thread per rank single-threads all of that; the guidance is now to divide the node's physical cores among the ranks exactly as for a CPU run, and to expect GPU utilisation below 100% because of the host phases rather than read it as a fault. Co-Authored-By: Claude Opus 5 (1M context) --- examples/README.md | 26 ++++++++++++++++++++++++-- 1 file changed, 24 insertions(+), 2 deletions(-) diff --git a/examples/README.md b/examples/README.md index 8835447..1fa7c09 100644 --- a/examples/README.md +++ b/examples/README.md @@ -39,9 +39,20 @@ import sbd # No init() needed — auto-initializes on first call result_cpu = sbd.tpb_diag(..., device='cpu') result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' -# result_omp = sbd.tpb_diag(..., device='gpu-omp') # NOT in the same process as 'cpu' +# result_omp = sbd.tpb_diag(..., device='gpu-omp') # do NOT mix with 'cpu' — see below ``` +**`'cpu'` and `'gpu-omp'` must not be used in the same process.** Both link the same +OpenMP runtime, and `_core_cpu` is built without offload support, so whichever loads +first initializes that runtime host-only. If it is the CPU backend, the OMP-offload +backend can no longer acquire a device and **silently runs its target regions on the +host**: correct energies, exit status 0, and the GPU sitting idle. There is no error to +catch, which is why it is worth knowing rather than discovering. `sbd.loaded_backends()` +reports what the current process has actually imported. + +`'cpu'` and `'gpu'` (Thrust) *can* share a process — Thrust does not route its device +work through OpenMP, so it has no equivalent interaction. + ## Available Test Data Paths below are written as the drivers use them, i.e. relative to a solver folder @@ -56,7 +67,18 @@ Smaller thresholds = more determinants = higher accuracy. **CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). -**GPU:** One MPI rank per GPU, `OMP_NUM_THREADS=1`. Each rank auto-assigned: `gpu_id = rank % num_gpus`. Use method 0 (matrix-free Davidson) for best GPU performance. +**GPU:** One MPI rank per GPU — each rank is auto-assigned `gpu_id = rank % num_gpus`. +Use method 0 (matrix-free Davidson) for best GPU performance. + +**Do not set `OMP_NUM_THREADS=1` for GPU runs.** One rank per GPU is about device +ownership, not thread count, and a GPU build still does real work on the host: helper +construction is host-threaded in both solvers (`tpb/helper.h`, `gdb/helper.h`), and for +GDB the heatbath expansion and carryover selection have no device implementation at all +(`gdb/expansion.h`, `gdb/carryover.h` carry no `thrust::` code and are compiled in +regardless of backend). One thread per rank single-threads all of it. Divide the node's +physical cores among the ranks exactly as for a CPU run — e.g. 96 cores with 8 ranks is +`OMP_NUM_THREADS=12`. Expect GPU utilisation below 100% as a result; that is the host +phases, not a fault. ## GDB (general determinant basis) From ed6fa0a212fcf4709ed0c878f2ae36e947b9797d Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 15:25:58 -0400 Subject: [PATCH 07/10] examples: drop the shared README; each solver folder stands alone An intermediate examples/README.md meant a reader looking for how to run something had to visit two files. Removed, with its content moved to where it is used rather than deleted: - Backend Selection (the --device values, available_backends()/loaded_backends(), and why 'cpu' and 'gpu-omp' must not share a process) -> examples/tpb/README.md - Available Test Data (the bundled h2o and n2 alpha lists) -> examples/tpb/README.md, whose drivers are what consume them - Performance Tips (thread counts for CPU and GPU runs) -> examples/tpb/README.md Its GDB paragraph is dropped rather than moved: the top README's Configuration section now explains the b_comm_size and t_comm_size constraints more fully, including whose limitation each one is, so the shorter version was redundant. References repointed: the top README indexes the folder directly, INSTALL.md's See Also, python/__init__.py's citation of the rank-to-device rule, and the python/examples signpost. Verified every relative link in all four files resolves and the sdist still ships the examples tree. Co-Authored-By: Claude Opus 5 (1M context) --- INSTALL.md | 4 +- README.md | 9 ++-- examples/README.md | 97 --------------------------------------- examples/tpb/README.md | 73 +++++++++++++++++++++++++++-- python/__init__.py | 2 +- python/examples/README.md | 10 ++-- 6 files changed, 80 insertions(+), 115 deletions(-) delete mode 100644 examples/README.md diff --git a/INSTALL.md b/INSTALL.md index 08e53a3..e2f1966 100644 --- a/INSTALL.md +++ b/INSTALL.md @@ -197,6 +197,6 @@ The Thrust backend is stamped too (`cuda:cc90`); the CPU backend reports `None`. ## See Also - [README](README.md) — what the package is, the API, and the SQD integration -- [`examples/README.md`](examples/README.md) — backend selection at runtime, bundled - test data and performance notes +- [`examples/tpb/README.md`](examples/tpb/README.md) — backend selection at runtime, + bundled test data and performance notes - [Troubleshooting](README.md#troubleshooting) — build and runtime symptoms diff --git a/README.md b/README.md index e9e4131..5ac24ca 100644 --- a/README.md +++ b/README.md @@ -51,14 +51,13 @@ built), building against an existing host MPI, and fuller verification. ## Examples -Located in [`examples/`](examples/README.md), organized by basis type since the -solvers take different subspaces and decompose over MPI differently. Each folder's -README is the authoritative list of what it contains and how to run it. +Located in `examples/`, organized by basis type since the solvers take different +subspaces and decompose over MPI differently. Each folder's README is the authoritative +guide to what it contains, how to run it, and the backend and threading settings that +matter for it. - [`examples/tpb/`](examples/tpb/README.md) — tensor-product basis: standalone TPB diagonalization, the SQD loops, and the subspace-enlargement driver. -- [`examples/README.md`](examples/README.md) — backend selection, `--device` values, - bundled test data and performance notes, shared by all examples. ## Integration with qiskit-addon-sqd diff --git a/examples/README.md b/examples/README.md deleted file mode 100644 index 1fa7c09..0000000 --- a/examples/README.md +++ /dev/null @@ -1,97 +0,0 @@ -# SBD examples - -Runtime behaviour shared by every example: which backend you get, what data is -bundled, and how to spend threads and GPUs. Solver-specific usage lives with the -examples themselves: - -- [`tpb/`](tpb/README.md) — TPB (tensor-product basis) and the SQD loops built on - it. The subspace is the Cartesian product of an alpha and a beta determinant list. - -- **Communication:** MPI for distributed computing -- **Backends:** CPU (host OpenMP, `--device cpu`), GPU (NVHPC Thrust, NVIDIA only, - `--device gpu`) and GPU (OpenMP target offload, NVIDIA and AMD, `--device - gpu-omp`), switchable at runtime via the `device` parameter - -Replace `--device gpu` with `--device gpu-omp` if you use AMD GPUs. - -## Backend Selection - -Every backend the toolchain supported was compiled into this one install, and -each is imported **lazily, on first use** — normally one per process. Select -per-call via `--device`: - -```bash ---device cpu # host OpenMP (default) ---device gpu # NVHPC Thrust (requires NVIDIA GPU + HPC SDK build) ---device gpu-omp # OpenMP target offload, NVIDIA and AMD GPUs ---device auto # GPU if available, else CPU -``` - -`sbd.available_backends()` reports what this install actually has (a static scan -— it does not import anything, so it is safe to call outside `mpirun`), and -`sbd.loaded_backends()` reports what the current process has pulled in. - -Within Python, backends can be selected per call — no re-initialization needed: - -```python -import sbd - -# No init() needed — auto-initializes on first call -result_cpu = sbd.tpb_diag(..., device='cpu') -result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' -# result_omp = sbd.tpb_diag(..., device='gpu-omp') # do NOT mix with 'cpu' — see below -``` - -**`'cpu'` and `'gpu-omp'` must not be used in the same process.** Both link the same -OpenMP runtime, and `_core_cpu` is built without offload support, so whichever loads -first initializes that runtime host-only. If it is the CPU backend, the OMP-offload -backend can no longer acquire a device and **silently runs its target regions on the -host**: correct energies, exit status 0, and the GPU sitting idle. There is no error to -catch, which is why it is worth knowing rather than discovering. `sbd.loaded_backends()` -reports what the current process has actually imported. - -`'cpu'` and `'gpu'` (Thrust) *can* share a process — Thrust does not route its device -work through OpenMP, so it has no equivalent interaction. - -## Available Test Data - -Paths below are written as the drivers use them, i.e. relative to a solver folder -such as `examples/tpb/`, which is how the drivers' own defaults are spelled. - -**H2O** (`../../vendor/sbd-upstream/data/h2o/`): `h2o-1em3` through `h2o-1em8` alpha determinant files. -**N2** (`../../vendor/sbd-upstream/data/n2/`): `1em3` through `1em7` and `3em4` through `3em7` alpha determinant files. - -Smaller thresholds = more determinants = higher accuracy. - -## Performance Tips - -**CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). - -**GPU:** One MPI rank per GPU — each rank is auto-assigned `gpu_id = rank % num_gpus`. -Use method 0 (matrix-free Davidson) for best GPU performance. - -**Do not set `OMP_NUM_THREADS=1` for GPU runs.** One rank per GPU is about device -ownership, not thread count, and a GPU build still does real work on the host: helper -construction is host-threaded in both solvers (`tpb/helper.h`, `gdb/helper.h`), and for -GDB the heatbath expansion and carryover selection have no device implementation at all -(`gdb/expansion.h`, `gdb/carryover.h` carry no `thrust::` code and are compiled in -regardless of backend). One thread per rank single-threads all of it. Divide the node's -physical cores among the ranks exactly as for a CPU run — e.g. 96 cores with 8 ranks is -`OMP_NUM_THREADS=12`. Expect GPU utilisation below 100% as a result; that is the host -phases, not a fault. - -## GDB (general determinant basis) - -`gdb_diag` spans the subspace with an explicit determinant list rather than the -Cartesian product TPB uses, and decomposes as `t_comm_size × b_comm_size × helper` -with its own field names. It has no example driver yet, so its decomposition is not -documented further here. - -`b_comm_size` must currently be 1: every rank passes the whole determinant list, so -splitting the basis communicator would have each rank diagonalize the full subspace -while believing it held a shard. - -## See Also - -- [Repository README](../README.md) — Installation, API reference -- [Upstream SBD library](https://github.com/r-ccs-cms/sbd) — C++ library overview diff --git a/examples/tpb/README.md b/examples/tpb/README.md index 5e0d15a..3efe7d3 100644 --- a/examples/tpb/README.md +++ b/examples/tpb/README.md @@ -4,9 +4,8 @@ Examples demonstrating SBD's capabilities for quantum chemistry calculations. ## Overview -These examples all use TPB, where the subspace is the Cartesian product of an -alpha and a beta determinant list. For backend selection, `--device` values and -the bundled test data, see [`../README.md`](../README.md). +These examples all use TPB, where the subspace is the Cartesian product of an alpha and +a beta determinant list. ## Examples @@ -341,8 +340,72 @@ When using more than one rank, specify at least `--adet_comm_size`. Examples: - **H2O**: ground state energy ≈ **-76.236 Hartree** - **N2**: ground state energy ≈ **-109.042 Hartree** (with 1e-3 dets) +## Backend Selection + +Every backend the toolchain supported was compiled into this one install, and +each is imported **lazily, on first use** — normally one per process. Select +per-call via `--device`: + +```bash +--device cpu # host OpenMP (default) +--device gpu # NVHPC Thrust (requires NVIDIA GPU + HPC SDK build) +--device gpu-omp # OpenMP target offload, NVIDIA and AMD GPUs +--device auto # GPU if available, else CPU +``` + +`sbd.available_backends()` reports what this install actually has (a static scan +— it does not import anything, so it is safe to call outside `mpirun`), and +`sbd.loaded_backends()` reports what the current process has pulled in. + +Within Python, backends can be selected per call — no re-initialization needed: + +```python +import sbd + +# No init() needed — auto-initializes on first call +result_cpu = sbd.tpb_diag(..., device='cpu') +result_gpu = sbd.tpb_diag(..., device='gpu') # fine alongside 'cpu' +# result_omp = sbd.tpb_diag(..., device='gpu-omp') # do NOT mix with 'cpu' — see below +``` + +**`'cpu'` and `'gpu-omp'` must not be used in the same process.** Both link the same +OpenMP runtime, and `_core_cpu` is built without offload support, so whichever loads +first initializes that runtime host-only. If it is the CPU backend, the OMP-offload +backend can no longer acquire a device and **silently runs its target regions on the +host**: correct energies, exit status 0, and the GPU sitting idle. There is no error to +catch, which is why it is worth knowing rather than discovering. `sbd.loaded_backends()` +reports what the current process has actually imported. + +`'cpu'` and `'gpu'` (Thrust) *can* share a process — Thrust does not route its device +work through OpenMP, so it has no equivalent interaction. + +## Available Test Data + +Paths below are written as the drivers use them, i.e. relative to a solver folder +such as `examples/tpb/`, which is how the drivers' own defaults are spelled. + +**H2O** (`../../vendor/sbd-upstream/data/h2o/`): `h2o-1em3` through `h2o-1em8` alpha determinant files. +**N2** (`../../vendor/sbd-upstream/data/n2/`): `1em3` through `1em7` and `3em4` through `3em7` alpha determinant files. + +Smaller thresholds = more determinants = higher accuracy. + +## Performance Tips + +**CPU:** Set `OMP_NUM_THREADS` to cores per MPI rank (e.g., 8 ranks × 4 threads = 32 cores). + +**GPU:** One MPI rank per GPU — each rank is auto-assigned `gpu_id = rank % num_gpus`. +Use method 0 (matrix-free Davidson) for best GPU performance. + +**Do not set `OMP_NUM_THREADS=1` for GPU runs.** One rank per GPU is about device +ownership, not thread count, and a GPU build still does real work on the host: helper +construction is host-threaded in both solvers (`tpb/helper.h`, `gdb/helper.h`), and for +GDB the heatbath expansion and carryover selection have no device implementation at all +(`gdb/expansion.h`, `gdb/carryover.h` carry no `thrust::` code and are compiled in +regardless of backend). One thread per rank single-threads all of it. Divide the node's +physical cores among the ranks exactly as for a CPU run — e.g. 96 cores with 8 ranks is +`OMP_NUM_THREADS=12`. Expect GPU utilisation below 100% as a result; that is the host +phases, not a fault. + ## See Also -- [`../README.md`](../README.md) — backend selection, bundled test data, - performance tips - [Repository README](../../README.md) — Installation, API reference diff --git a/python/__init__.py b/python/__init__.py index f9dee30..d74a2fe 100644 --- a/python/__init__.py +++ b/python/__init__.py @@ -465,7 +465,7 @@ def get_device_id(device=None): Uses the communicator rank, i.e. the same ``gpu_id = rank % num_gpus`` convention SBD's own diagonalization applies, documented in - ``examples/README.md``. + ``examples/tpb/README.md``. """ _ensure_initialized() return get_backend(device).planned_device_id(get_comm()) diff --git a/python/examples/README.md b/python/examples/README.md index 76d65e8..e78af36 100644 --- a/python/examples/README.md +++ b/python/examples/README.md @@ -1,10 +1,10 @@ # Moved -The examples now live at the top level, outside the installed package: +The examples now live at the top level, outside the installed package, grouped by basis +type: -- [`examples/tpb/`](../../examples/tpb/README.md) — TPB (tensor-product basis) and - the SQD loops, i.e. everything that used to be in this directory. -- [`examples/README.md`](../../examples/README.md) — backend selection, `--device` - values, bundled test data and performance notes, shared by all examples. +- [`examples/tpb/`](../../examples/tpb/README.md) — TPB (tensor-product basis) and the + SQD loops, i.e. everything that used to be in this directory. Its README also covers + backend selection, the bundled test data and threading. This file is a signpost for older links and can be removed once they have aged out. From 00d16cc3a62af632aea9993b1476e8074d6559d2 Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Tue, 29 Sep 2026 16:12:35 -0400 Subject: [PATCH 08/10] test: point COUNTS_PATH at the moved counts file PR #18 landed test/test_sqd_integration.py while this branch was open. Its conftest resolves the curated h2o counts under python/examples/, which this branch moves to examples/tpb/. Neither change conflicts textually, so the merge was clean and the merged tree still broke -- every CI job failed on FileNotFoundError: .../python/examples/count_dict_h2o.json which is exactly what the comment above COUNTS_PATH says should happen when a move does not update it. Retarget the constant and drop the MANIFEST.in line for the shared examples README this branch removed. Co-Authored-By: Claude Opus 5 (1M context) --- MANIFEST.in | 1 - test/conftest.py | 4 +--- 2 files changed, 1 insertion(+), 4 deletions(-) diff --git a/MANIFEST.in b/MANIFEST.in index 6bb3f07..ff153aa 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -17,5 +17,4 @@ include vendor/sbd-upstream/LICENSE.txt # Example scripts and notebook, referenced by the README. recursive-include examples *.py *.ipynb *.md *.json -include python/examples/README.md include INSTALL.md diff --git a/test/conftest.py b/test/conftest.py index 159a06a..49040d2 100644 --- a/test/conftest.py +++ b/test/conftest.py @@ -28,9 +28,7 @@ # So a missing file means a broken checkout or a move that did not update this file, and # the tests should fail and say so rather than skip and report success. DATA_DIR = Path(__file__).resolve().parents[1] / "vendor" / "sbd-upstream" / "data" -COUNTS_PATH = ( - Path(__file__).resolve().parents[1] / "python" / "examples" / "count_dict_h2o.json" -) +COUNTS_PATH = Path(__file__).resolve().parents[1] / "examples" / "tpb" / "count_dict_h2o.json" # Slow tests are opt-in through a command-line flag rather than excluded by default, From 1df366c77e2d553bc86ce21887546cc8753bbe0d Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Wed, 30 Sep 2026 09:55:25 -0400 Subject: [PATCH 09/10] setup.py: find nvc++ under NVHPC_ROOT and under compilers/, not just /bin Detection probed exactly one path, $NVHPC_HOME/bin/nvc++. NVHPC does not put the compiler there: the version root holds compilers/ cuda/ math_libs/ comm_libs/ and the binary is at /compilers/bin/nvc++. NVIDIA's own modulefile does setenv NVHPC_ROOT $nvhome/$target/$version prepend-path PATH $nvcompdir/bin so NVHPC_ROOT names the version root, one level above the compiler. Setting NVHPC_HOME to that root -- the natural guess, and what copying $NVHPC_ROOT gives you -- found nothing, and NVHPC_ROOT was never consulted. Neither failure is loud. SBD_BUILD_BACKEND=auto falls back to a CPU-only build, pip hides setup.py's output without -v, and the install exits 0; the user finds out later from "Device 'gpu' requested but its backend is not usable". Verified on a real NVHPC 26.3 tree with nvc++ off PATH: before, both NVHPC_HOME= and NVHPC_ROOT= gave "will build CPU backend only"; after, both give "will build CPU, Thrust GPU and OMP-offload GPU backends". Read NVHPC_ROOT as well as NVHPC_HOME, and under each try bin/ and compilers/bin/, so the compilers directory, the version root, and a bare `module load nvhpc` all work. Also compare PATH entries rather than substrings when deciding whether the directory is already there. The "Found NVIDIA HPC SDK at: " prefix is kept because the troubleshooting section quotes it; the variable name is appended. test/test_build_detection.py is new -- setup.py had no tests. It extracts the function rather than importing setup.py, which would run a build, and covers both variables against both layouts, a stale NVHPC_HOME not masking a good NVHPC_ROOT, and a negative control so the suite cannot pass against a function that always claims success. Co-Authored-By: Claude Opus 5 (1M context) --- INSTALL.md | 3 + README.md | 4 +- setup.py | 47 +++++++++---- test/test_build_detection.py | 133 +++++++++++++++++++++++++++++++++++ 4 files changed, 174 insertions(+), 13 deletions(-) create mode 100644 test/test_build_detection.py diff --git a/INSTALL.md b/INSTALL.md index e2f1966..e1ea188 100644 --- a/INSTALL.md +++ b/INSTALL.md @@ -91,6 +91,9 @@ pip install -e . --no-build-isolation --force-reinstall --no-deps ```bash # --- NVIDIA GPU backends (Thrust and OpenMP target-offload): point at NVHPC. # Only needed if nvc++ is not already on PATH. Adjust the path. +# Either the compilers directory or the version root above it works, and +# NVHPC_ROOT -- which NVIDIA's own modulefile sets to the version root -- is +# read too, so `module load nvhpc` alone is enough. export NVHPC_HOME=/opt/nvidia/hpc_sdk/Linux_x86_64/2025/compilers # --- AMD GPU backend (OpenMP target-offload): point at ROCm LLVM toolchain diff --git a/README.md b/README.md index 5ac24ca..e091a44 100644 --- a/README.md +++ b/README.md @@ -265,7 +265,9 @@ The optional `device` parameter overrides the default set by `init()`. the build respects a caller-set compiler, so `nvc++`/`amdclang++` never run. Unset `CC`/`CXX`, or keep conda compilers out of the build env. -**GPU not building:** On NVIDIA check `which nvc++` and set `NVHPC_HOME`. On AMD +**GPU not building:** On NVIDIA check `which nvc++` and set `NVHPC_HOME` (or rely on +`NVHPC_ROOT` from `module load nvhpc`; either the compilers directory or the version +root works). On AMD check `which amdclang++` and set `ROCM_HOME`. The build prints which toolchain it picked (`Found amdclang++ in PATH: …` / `Found NVIDIA HPC SDK at: …`) and, for the offload backend, the resolved architecture; on a host with both toolchains diff --git a/setup.py b/setup.py index 63bff45..394d1a9 100644 --- a/setup.py +++ b/setup.py @@ -473,19 +473,42 @@ def _homebrew_prefix(): return None +# Where nvc++ sits relative to whichever directory the environment points at. +# NVHPC installs the compiler in /compilers/bin, so probing only +# /bin misses the value NVIDIA itself hands out -- see NVHPC_VARS below. +_NVHPC_BIN_RELATIVE = ('bin', os.path.join('compilers', 'bin')) + +# NVHPC_HOME is this package's own variable and is documented in INSTALL.md as +# the compilers directory. NVHPC_ROOT is set by NVIDIA's shipped modulefile, to +# the version root .../Linux_x86_64/ -- that file does +# setenv NVHPC_ROOT $nvhome/$target/$version +# prepend-path PATH $nvcompdir/bin +# so the compiler is one level down from NVHPC_ROOT. Accept both variables, and +# under each accept both layouts, so that the compilers directory, the version +# root, and a copy of $NVHPC_ROOT all work rather than silently yielding a +# CPU-only build. +_NVHPC_VARS = ('NVHPC_HOME', 'NVHPC_ROOT') + + def find_nvidia_hpc_sdk(): - nvhpc_home = os.environ.get('NVHPC_HOME', None) - if nvhpc_home: - nvcxx_path = os.path.join(nvhpc_home, 'bin', 'nvc++') - if os.path.exists(nvcxx_path): - print(f"Found NVIDIA HPC SDK at: {nvhpc_home}") - nvhpc_bin = os.path.join(nvhpc_home, 'bin') - current_path = os.environ.get('PATH', '') - if nvhpc_bin not in current_path: - os.environ['PATH'] = f"{nvhpc_bin}:{current_path}" - return nvcxx_path, True - else: - print(f"Warning: NVHPC_HOME set to {nvhpc_home} but nvc++ not found") + """Locate nvc++: the NVHPC_* variables first, then PATH.""" + for var in _NVHPC_VARS: + root = os.environ.get(var) + if not root: + continue + for relative in _NVHPC_BIN_RELATIVE: + nvcxx_path = os.path.join(root, relative, 'nvc++') + if os.path.exists(nvcxx_path): + print(f"Found NVIDIA HPC SDK at: {nvcxx_path} (via {var})") + nvhpc_bin = os.path.dirname(nvcxx_path) + current_path = os.environ.get('PATH', '') + # Compare path entries, not substrings: a directory whose name + # happens to be contained in some other entry is not on PATH. + if nvhpc_bin not in current_path.split(os.pathsep): + os.environ['PATH'] = f"{nvhpc_bin}{os.pathsep}{current_path}" + return nvcxx_path, True + tried = ', '.join(os.path.join(root, r, 'nvc++') for r in _NVHPC_BIN_RELATIVE) + print(f"Warning: {var} is set but no nvc++ found. Tried: {tried}") import shutil nvcxx_path = shutil.which('nvc++') if nvcxx_path: diff --git a/test/test_build_detection.py b/test/test_build_detection.py new file mode 100644 index 0000000..7f602ab --- /dev/null +++ b/test/test_build_detection.py @@ -0,0 +1,133 @@ +# This code is a Qiskit project. +# +# (C) Copyright IBM 2026. +# +# This code is licensed under the Apache License, Version 2.0. You may +# obtain a copy of this license in the LICENSE.txt file in the root directory +# of this source tree or at http://www.apache.org/licenses/LICENSE-2.0. +# +# Any modifications or derivative works of this code must retain this +# copyright notice, and modified files need to carry a notice indicating +# that they have been altered from the originals. + +"""Check that the build finds a GPU toolchain wherever the environment points. + +A missed toolchain is not a loud failure: ``SBD_BUILD_BACKEND=auto`` falls back +to a CPU-only build, pip hides setup.py's output unless ``-v`` is passed, and the +install then succeeds. The user learns about it later, from +``Device 'gpu' requested but its backend is not usable``. So the detection is +worth a test of its own. + +The case that motivated this: NVHPC keeps ``nvc++`` in +``/compilers/bin``, while ``NVHPC_ROOT`` -- set by NVIDIA's own +modulefile -- points at the version root. Probing only ``/bin`` therefore +misses the path NVIDIA hands out. + +``setup.py`` runs its detection at import time and cannot be imported without +triggering a build, so the functions under test are extracted from its source. +""" + +from __future__ import annotations + +import ast +import os +import pathlib + +import pytest + +SETUP_PY = pathlib.Path(__file__).resolve().parents[1] / "setup.py" + +# Module-level names the extracted function closes over. +_CONSTANTS = {"_NVHPC_BIN_RELATIVE", "_NVHPC_VARS"} +_FUNCTIONS = {"find_nvidia_hpc_sdk"} + + +def _load_from_setup_py(): + """Pull the detection helpers out of setup.py without executing the rest.""" + if not SETUP_PY.is_file(): + pytest.skip(f"setup.py not found at {SETUP_PY}") + namespace: dict = {"os": os} + tree = ast.parse(SETUP_PY.read_text()) + for node in tree.body: + keep = ( + isinstance(node, ast.Assign) + and any(getattr(t, "id", None) in _CONSTANTS for t in node.targets) + ) or (isinstance(node, ast.FunctionDef) and node.name in _FUNCTIONS) + if keep: + exec(compile(ast.Module([node], []), str(SETUP_PY), "exec"), namespace) + missing = (_CONSTANTS | _FUNCTIONS) - namespace.keys() + assert not missing, f"setup.py no longer defines {sorted(missing)}" + return namespace["find_nvidia_hpc_sdk"] + + +@pytest.fixture +def fake_nvhpc(tmp_path): + """An NVHPC tree with the real layout: nvc++ under /compilers/bin.""" + root = tmp_path / "Linux_x86_64" / "26.3" + binary = root / "compilers" / "bin" / "nvc++" + binary.parent.mkdir(parents=True) + binary.write_text("#!/bin/sh\n") + return root + + +@pytest.fixture +def clean_env(monkeypatch): + """No NVHPC_* variables, and a PATH with no nvc++ on it.""" + for name in ("NVHPC_HOME", "NVHPC_ROOT"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setenv("PATH", os.pathsep.join(("/usr/bin", "/bin"))) + + +@pytest.mark.parametrize( + "variable, points_at", + [ + ("NVHPC_HOME", "compilers"), + ("NVHPC_HOME", "root"), + ("NVHPC_ROOT", "compilers"), + ("NVHPC_ROOT", "root"), + ], + ids=["home-compilers", "home-root", "root-compilers", "root-root"], +) +def test_finds_nvcxx_from_either_variable_and_layout( + variable, points_at, fake_nvhpc, clean_env, monkeypatch +): + """Both variables, and both the version root and its compilers/ subdirectory. + + ``NVHPC_HOME`` pointed at the version root used to fail, and ``NVHPC_ROOT`` + was not consulted at all -- each silently yielding a CPU-only build on a + machine with a working compiler. + """ + find = _load_from_setup_py() + target = fake_nvhpc if points_at == "root" else fake_nvhpc / "compilers" + monkeypatch.setenv(variable, str(target)) + + path, found = find() + + assert found, f"{variable} pointing at the {points_at} was not resolved" + assert pathlib.Path(path) == fake_nvhpc / "compilers" / "bin" / "nvc++" + assert os.path.dirname(path) in os.environ["PATH"].split(os.pathsep), ( + "the compiler was found but its directory was not prepended to PATH" + ) + + +def test_reports_nothing_found_when_there_is_nothing(clean_env): + """The negative control: no variables, no nvc++ on PATH, no claim of one. + + Without this, a test that only ever asserts success would pass against a + function that always returned a path. + """ + find = _load_from_setup_py() + path, found = find() + assert not found and path is None + + +def test_a_wrong_variable_does_not_mask_a_right_one(fake_nvhpc, clean_env, monkeypatch): + """A stale NVHPC_HOME must not stop NVHPC_ROOT from being used.""" + find = _load_from_setup_py() + monkeypatch.setenv("NVHPC_HOME", str(fake_nvhpc / "does-not-exist")) + monkeypatch.setenv("NVHPC_ROOT", str(fake_nvhpc)) + + path, found = find() + + assert found, "a bad NVHPC_HOME shadowed a good NVHPC_ROOT" + assert pathlib.Path(path) == fake_nvhpc / "compilers" / "bin" / "nvc++" From 9df759aa6d0c9fa8e4cb19dd3ef34b59c46d1fa9 Mon Sep 17 00:00:00 2001 From: Sophia Wen Date: Wed, 30 Sep 2026 13:43:13 -0400 Subject: [PATCH 10/10] test: drop test_build_detection.py, out of scope for this PR Flagged in review as unrelated to the PR description, which it is -- this branch moves examples and splits the install docs. The setup.py NVHPC_ROOT fix stays, since INSTALL.md documents the variable and the two belong together, but its test does not need to ride along here. Co-Authored-By: Claude Opus 5 (1M context) --- test/test_build_detection.py | 133 ----------------------------------- 1 file changed, 133 deletions(-) delete mode 100644 test/test_build_detection.py diff --git a/test/test_build_detection.py b/test/test_build_detection.py deleted file mode 100644 index 7f602ab..0000000 --- a/test/test_build_detection.py +++ /dev/null @@ -1,133 +0,0 @@ -# This code is a Qiskit project. -# -# (C) Copyright IBM 2026. -# -# This code is licensed under the Apache License, Version 2.0. You may -# obtain a copy of this license in the LICENSE.txt file in the root directory -# of this source tree or at http://www.apache.org/licenses/LICENSE-2.0. -# -# Any modifications or derivative works of this code must retain this -# copyright notice, and modified files need to carry a notice indicating -# that they have been altered from the originals. - -"""Check that the build finds a GPU toolchain wherever the environment points. - -A missed toolchain is not a loud failure: ``SBD_BUILD_BACKEND=auto`` falls back -to a CPU-only build, pip hides setup.py's output unless ``-v`` is passed, and the -install then succeeds. The user learns about it later, from -``Device 'gpu' requested but its backend is not usable``. So the detection is -worth a test of its own. - -The case that motivated this: NVHPC keeps ``nvc++`` in -``/compilers/bin``, while ``NVHPC_ROOT`` -- set by NVIDIA's own -modulefile -- points at the version root. Probing only ``/bin`` therefore -misses the path NVIDIA hands out. - -``setup.py`` runs its detection at import time and cannot be imported without -triggering a build, so the functions under test are extracted from its source. -""" - -from __future__ import annotations - -import ast -import os -import pathlib - -import pytest - -SETUP_PY = pathlib.Path(__file__).resolve().parents[1] / "setup.py" - -# Module-level names the extracted function closes over. -_CONSTANTS = {"_NVHPC_BIN_RELATIVE", "_NVHPC_VARS"} -_FUNCTIONS = {"find_nvidia_hpc_sdk"} - - -def _load_from_setup_py(): - """Pull the detection helpers out of setup.py without executing the rest.""" - if not SETUP_PY.is_file(): - pytest.skip(f"setup.py not found at {SETUP_PY}") - namespace: dict = {"os": os} - tree = ast.parse(SETUP_PY.read_text()) - for node in tree.body: - keep = ( - isinstance(node, ast.Assign) - and any(getattr(t, "id", None) in _CONSTANTS for t in node.targets) - ) or (isinstance(node, ast.FunctionDef) and node.name in _FUNCTIONS) - if keep: - exec(compile(ast.Module([node], []), str(SETUP_PY), "exec"), namespace) - missing = (_CONSTANTS | _FUNCTIONS) - namespace.keys() - assert not missing, f"setup.py no longer defines {sorted(missing)}" - return namespace["find_nvidia_hpc_sdk"] - - -@pytest.fixture -def fake_nvhpc(tmp_path): - """An NVHPC tree with the real layout: nvc++ under /compilers/bin.""" - root = tmp_path / "Linux_x86_64" / "26.3" - binary = root / "compilers" / "bin" / "nvc++" - binary.parent.mkdir(parents=True) - binary.write_text("#!/bin/sh\n") - return root - - -@pytest.fixture -def clean_env(monkeypatch): - """No NVHPC_* variables, and a PATH with no nvc++ on it.""" - for name in ("NVHPC_HOME", "NVHPC_ROOT"): - monkeypatch.delenv(name, raising=False) - monkeypatch.setenv("PATH", os.pathsep.join(("/usr/bin", "/bin"))) - - -@pytest.mark.parametrize( - "variable, points_at", - [ - ("NVHPC_HOME", "compilers"), - ("NVHPC_HOME", "root"), - ("NVHPC_ROOT", "compilers"), - ("NVHPC_ROOT", "root"), - ], - ids=["home-compilers", "home-root", "root-compilers", "root-root"], -) -def test_finds_nvcxx_from_either_variable_and_layout( - variable, points_at, fake_nvhpc, clean_env, monkeypatch -): - """Both variables, and both the version root and its compilers/ subdirectory. - - ``NVHPC_HOME`` pointed at the version root used to fail, and ``NVHPC_ROOT`` - was not consulted at all -- each silently yielding a CPU-only build on a - machine with a working compiler. - """ - find = _load_from_setup_py() - target = fake_nvhpc if points_at == "root" else fake_nvhpc / "compilers" - monkeypatch.setenv(variable, str(target)) - - path, found = find() - - assert found, f"{variable} pointing at the {points_at} was not resolved" - assert pathlib.Path(path) == fake_nvhpc / "compilers" / "bin" / "nvc++" - assert os.path.dirname(path) in os.environ["PATH"].split(os.pathsep), ( - "the compiler was found but its directory was not prepended to PATH" - ) - - -def test_reports_nothing_found_when_there_is_nothing(clean_env): - """The negative control: no variables, no nvc++ on PATH, no claim of one. - - Without this, a test that only ever asserts success would pass against a - function that always returned a path. - """ - find = _load_from_setup_py() - path, found = find() - assert not found and path is None - - -def test_a_wrong_variable_does_not_mask_a_right_one(fake_nvhpc, clean_env, monkeypatch): - """A stale NVHPC_HOME must not stop NVHPC_ROOT from being used.""" - find = _load_from_setup_py() - monkeypatch.setenv("NVHPC_HOME", str(fake_nvhpc / "does-not-exist")) - monkeypatch.setenv("NVHPC_ROOT", str(fake_nvhpc)) - - path, found = find() - - assert found, "a bad NVHPC_HOME shadowed a good NVHPC_ROOT" - assert pathlib.Path(path) == fake_nvhpc / "compilers" / "bin" / "nvc++"