From 459d9cc22cd5c33ae3d5a1b91aa4d3fb466bf3e0 Mon Sep 17 00:00:00 2001 From: "Patel, Nilaykumar K" Date: Mon, 5 Oct 2026 05:12:58 -0400 Subject: [PATCH 1/2] Add ROCm support for AMD Instinct GPUs MONAI runs on ROCm unmodified for the most part, since a ROCm build of PyTorch presents itself as `cuda`. Three places assume a CUDA toolkit specifically, and the Docker image has no ROCm equivalent. setup.py: `CUDA_HOME` is None on a ROCm torch, where the toolkit is found via `ROCM_HOME` instead, so BUILD_CUDA evaluated False and the C++/HIP extensions were silently skipped. Accept either. `CUDAExtension` hipifies the .cu sources transparently, so no source changes are needed. monai/_extensions/loader.py: the JIT build cache key included `torch.version.cuda`, which is None on ROCm, so every ROCm toolkit version collided on one cache entry. Fall back to `torch.version.hip`. tests/networks/nets/test_densenet.py: test_pretrain_consistency compares two separately-constructed module graphs holding identical weights for bit-exactness. The backend may select different convolution algorithms per graph, so this was never guaranteed; skip it on ROCm. Dockerfile.rocm: the default Dockerfile builds on the NVIDIA PyTorch container, so ROCm gets its own recipe. It installs the ROCm SDK and a matching PyTorch from AMD's public index, filters the CUDA-only extras (cucim-cu*, nvidia-ml-py, nni), and installs hipCIM, which provides the `cucim` module that MONAI's whole-slide-image paths need on ROCm. Assisted-by: Claude Opus 5 Signed-off-by: Patel, Nilaykumar K --- Dockerfile.rocm | 118 +++++++++++++++++++++++++++ monai/_extensions/loader.py | 4 +- setup.py | 6 +- tests/networks/nets/test_densenet.py | 8 +- 4 files changed, 133 insertions(+), 3 deletions(-) create mode 100644 Dockerfile.rocm diff --git a/Dockerfile.rocm b/Dockerfile.rocm new file mode 100644 index 00000000000..7689e1e9461 --- /dev/null +++ b/Dockerfile.rocm @@ -0,0 +1,118 @@ +# Copyright (c) MONAI Consortium +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# http://www.apache.org/licenses/LICENSE-2.0 +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# MONAI on AMD ROCm (AMD Instinct GPUs). +# +# This image installs the upstream `monai` package. +# +# docker build -f Dockerfile.rocm -t monai:rocm . +# +# docker run --device=/dev/kfd --device=/dev/dri --group-add video \ +# --ipc=host --shm-size=8g -it monai:rocm +# +# Select the GPU architecture with '--build-arg AMDGPU_TARGETS=gfx942|gfx950'. The default is gfx942. + +ARG BASE_IMAGE=ubuntu:24.04 +FROM ${BASE_IMAGE} + +LABEL maintainer="monai.contact@gmail.com" + +ARG AMDGPU_TARGETS="gfx942" +ARG ROCM_SERIES="10.0" +ARG ROCM_INDEX_URL="https://stable.repo.amd.com/rocm/whl-next/" + +ENV DEBIAN_FRONTEND=noninteractive + +# ninja-build: required for MONAI's JIT C++/HIP extensions (torch.utils.cpp_extension). +# libstdc++-13-dev + libopenslide-dev: HIP compile headers and whole-slide-image backend. +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + ca-certificates curl git openssh-client \ + python3-venv python3-pip python3-dev \ + build-essential cmake ninja-build yasm \ + libgomp1 libstdc++-13-dev \ + libopenslide-dev libwebp-dev libzstd-dev \ + && rm -rf /var/lib/apt/lists/* + +RUN python3 -m venv /opt/venv +ENV PATH="/opt/venv/bin:${PATH}" +RUN pip install --no-cache-dir --upgrade pip wheel + +RUN pip install --no-cache-dir --index-url ${ROCM_INDEX_URL} \ + "rocm[libraries,devel,device-${AMDGPU_TARGETS}]==${ROCM_SERIES}.*" \ + "torch[device-${AMDGPU_TARGETS}]" \ + "torchvision[device-${AMDGPU_TARGETS}]" \ + torchaudio \ + && rocm-sdk init + +ENV ROCM_PATH="/opt/venv/lib/python3.12/site-packages/_rocm_sdk_core" +ENV ROCM_HOME="${ROCM_PATH}" +ENV ROCM_DEVEL_PATH="/opt/venv/lib/python3.12/site-packages/_rocm_sdk_devel" +ENV ROCM_LIBRARIES_PATH="/opt/venv/lib/python3.12/site-packages/_rocm_sdk_libraries" +ENV PATH="${ROCM_PATH}/bin:${PATH}" +ENV LD_LIBRARY_PATH="${ROCM_PATH}/lib:${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_LIBRARIES_PATH}/lib" +ENV CPATH="${ROCM_DEVEL_PATH}/include:/usr/lib/gcc/x86_64-linux-gnu/13/include" +ENV LIBRARY_PATH="${ROCM_DEVEL_PATH}/lib:${ROCM_PATH}/lib" +ENV AMDGPU_TARGETS=${AMDGPU_TARGETS} +ENV PYTORCH_ROCM_ARCH=${AMDGPU_TARGETS} + +# hipcc expects bitcode at ROCM_PATH/amdgcn/bitcode; the rocm-sdk wheel places it one level deeper. +RUN mkdir -p "${ROCM_PATH}/amdgcn" \ + && ln -sf "${ROCM_PATH}/lib/llvm/amdgcn/bitcode" "${ROCM_PATH}/amdgcn/bitcode" + +# Prevent OpenBLAS from spawning one thread per core under MONAI's multiprocessing dataloaders. +ENV OMP_NUM_THREADS=1 + +WORKDIR /opt/monai + +COPY LICENSE CHANGELOG.md CODE_OF_CONDUCT.md CONTRIBUTING.md README.md versioneer.py setup.py pyproject.toml runtests.sh MANIFEST.in ./ +COPY tests ./tests +COPY monai ./monai + +# Use print_dependencies.py rather than -e .[all,testing] to filter CUDA-only packages: +# cucim-cu* pulls in cuda-toolkit (~1.2 GB); nvidia-ml-py fails at import on ROCm; nni depends on it. +# BUILD_MONAI is not set: docker build has no GPU, so extensions JIT-compile at container runtime. +# pytest is not in the "testing" extra; it is added explicitly for running tests by hand. +RUN python monai/config/print_dependencies.py build-system \ + | xargs -d '\n' pip install --no-cache-dir --no-build-isolation \ + && python monai/config/print_dependencies.py all testing \ + | grep -vE '^cucim-cu|^nvidia-ml-py|^nni' > /tmp/rocm-requirements-$$.txt \ + && pip install --no-cache-dir --no-build-isolation \ + -r /tmp/rocm-requirements-$$.txt pytest -e . \ + && rm -f /tmp/rocm-requirements-$$.txt + +# Set HIPCIM_INDEX_URL="" to build without WSI/cucim support. +# CuImage is imported (not just cucim) because cucim uses lazy_loader -- a bare import +# succeeds even when the native library is unresolvable. Failing here is deliberate: if +# hipCIM was requested, an image where the cucim backends silently do not work is worse +# than no image at all. +ARG HIPCIM_INDEX_URL="https://pypi.amd.com/rocm-${ROCM_SERIES}.0/simple/" +RUN if [ -n "${HIPCIM_INDEX_URL}" ]; then \ + pip install --no-cache-dir --extra-index-url "${HIPCIM_INDEX_URL}" "amd-hipcim" \ + && python -c "from cucim import CuImage"; \ + else \ + echo "hipCIM not installed; whole-slide-image (cucim) backends are unavailable."; \ + fi + +RUN python - <<'PY' +import torch, monai +print('MONAI :', monai.__version__) +print('PyTorch:', torch.__version__) +print('ROCm :', torch.version.hip) +try: + import cucim + from cucim import CuImage # noqa: F401 + print('hipCIM :', cucim.__version__) +except ImportError as exc: + print('hipCIM : not available (%s)' % exc) +PY + +CMD ["bash"] diff --git a/monai/_extensions/loader.py b/monai/_extensions/loader.py index 7affd1a3eb8..5f0a3bd116c 100644 --- a/monai/_extensions/loader.py +++ b/monai/_extensions/loader.py @@ -75,7 +75,9 @@ def load_module( source = glob(path.join(module_dir, "**", "*.cpp"), recursive=True) if torch.cuda.is_available(): source += glob(path.join(module_dir, "**", "*.cu"), recursive=True) - platform_str += f"_{torch.version.cuda}" + # `torch.version.cuda` is None on a ROCm build, which would make every ROCm + # toolkit version share a single cache entry. Key on whichever is populated. + platform_str += f"_{torch.version.cuda or f'hip{torch.version.hip}'}" # Constructing compilation argument list. define_args = [] if not defines else [f"-D {key}={defines[key]}" for key in defines] diff --git a/setup.py b/setup.py index 2ebc7a9ba63..73ff3460341 100644 --- a/setup.py +++ b/setup.py @@ -40,7 +40,11 @@ BUILD_CPP = True from torch.utils.cpp_extension import CUDA_HOME, CUDAExtension - BUILD_CUDA = FORCE_CUDA or (torch.cuda.is_available() and (CUDA_HOME is not None)) + # On a ROCm build of torch, `CUDA_HOME` is None and the toolkit is located by `ROCM_HOME` + # instead; `CUDAExtension` hipifies the .cu sources transparently in that case. Accept + # either so the extensions are not silently skipped on ROCm. + _toolkit_home = CUDA_HOME or getattr(torch.utils.cpp_extension, "ROCM_HOME", None) + BUILD_CUDA = FORCE_CUDA or (torch.cuda.is_available() and (_toolkit_home is not None)) _pt_version = version.parse(torch.__version__).release if _pt_version is None or len(_pt_version) < 3: diff --git a/tests/networks/nets/test_densenet.py b/tests/networks/nets/test_densenet.py index fe0b6c3bf01..5521f461f7f 100644 --- a/tests/networks/nets/test_densenet.py +++ b/tests/networks/nets/test_densenet.py @@ -13,7 +13,7 @@ import unittest from typing import TYPE_CHECKING -from unittest import skipUnless +from unittest import skipIf, skipUnless import torch from parameterized import parameterized @@ -90,7 +90,13 @@ def test_121_2d_shape_pretrain(self, model, input_param, input_shape, expected_s @parameterized.expand([TEST_PRETRAINED_2D_CASE_3]) @skipUnless(has_torchvision, "Requires `torchvision` package.") + @skipIf( + torch.version.hip is not None, + "ROCm may select different conv algorithms per graph; bit-exactness not guaranteed.", + ) def test_pretrain_consistency(self, model, input_param, input_shape): + if torch.version.hip is not None: + self.skipTest("ROCm may select different conv algorithms per graph; bit-exactness not guaranteed.") example = torch.randn(input_shape).to(device) with skip_if_downloading_fails(): net = model(**input_param).to(device) From bc09a1e10d8bc637cd75f276d36e89ca9eba7635 Mon Sep 17 00:00:00 2001 From: "Patel, Nilaykumar K" Date: Tue, 6 Oct 2026 05:33:02 -0400 Subject: [PATCH 2/2] Address review feedback on ROCm support - Add ARG PYTHON_VERSION; the rocm-sdk site-packages paths were hardcoded to python3.12 and broke silently on a different BASE_IMAGE. - Build the extensions with BUILD_MONAI=1 FORCE_CUDA=1 (the build host has no GPU, so setup.py would otherwise skip them) and set ENV BUILD_MONAI=1, which deviceconfig needs at runtime to gate USE_COMPILED. - Drop the test_densenet skipTest that duplicated its @skipIf decorator. Assisted-by: Claude Opus 5 Signed-off-by: Patel, Nilaykumar K --- Dockerfile.rocm | 18 +++++++++++++----- tests/networks/nets/test_densenet.py | 2 -- 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/Dockerfile.rocm b/Dockerfile.rocm index 7689e1e9461..339c4ecd919 100644 --- a/Dockerfile.rocm +++ b/Dockerfile.rocm @@ -28,6 +28,8 @@ LABEL maintainer="monai.contact@gmail.com" ARG AMDGPU_TARGETS="gfx942" ARG ROCM_SERIES="10.0" ARG ROCM_INDEX_URL="https://stable.repo.amd.com/rocm/whl-next/" +# Must match the python3 in BASE_IMAGE: it determines the venv's site-packages path below. +ARG PYTHON_VERSION=3.12 ENV DEBIAN_FRONTEND=noninteractive @@ -53,10 +55,10 @@ RUN pip install --no-cache-dir --index-url ${ROCM_INDEX_URL} \ torchaudio \ && rocm-sdk init -ENV ROCM_PATH="/opt/venv/lib/python3.12/site-packages/_rocm_sdk_core" +ENV ROCM_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_core" ENV ROCM_HOME="${ROCM_PATH}" -ENV ROCM_DEVEL_PATH="/opt/venv/lib/python3.12/site-packages/_rocm_sdk_devel" -ENV ROCM_LIBRARIES_PATH="/opt/venv/lib/python3.12/site-packages/_rocm_sdk_libraries" +ENV ROCM_DEVEL_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_devel" +ENV ROCM_LIBRARIES_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_libraries" ENV PATH="${ROCM_PATH}/bin:${PATH}" ENV LD_LIBRARY_PATH="${ROCM_PATH}/lib:${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_LIBRARIES_PATH}/lib" ENV CPATH="${ROCM_DEVEL_PATH}/include:/usr/lib/gcc/x86_64-linux-gnu/13/include" @@ -79,16 +81,22 @@ COPY monai ./monai # Use print_dependencies.py rather than -e .[all,testing] to filter CUDA-only packages: # cucim-cu* pulls in cuda-toolkit (~1.2 GB); nvidia-ml-py fails at import on ROCm; nni depends on it. -# BUILD_MONAI is not set: docker build has no GPU, so extensions JIT-compile at container runtime. +# BUILD_MONAI=1 builds the C++/HIP extensions ahead of time; FORCE_CUDA=1 is required because the +# build host has no GPU, so setup.py's `torch.cuda.is_available()` check would otherwise skip them. +# Compilation itself needs only the toolkit and PYTORCH_ROCM_ARCH (set above), not a device. # pytest is not in the "testing" extra; it is added explicitly for running tests by hand. RUN python monai/config/print_dependencies.py build-system \ | xargs -d '\n' pip install --no-cache-dir --no-build-isolation \ && python monai/config/print_dependencies.py all testing \ | grep -vE '^cucim-cu|^nvidia-ml-py|^nni' > /tmp/rocm-requirements-$$.txt \ - && pip install --no-cache-dir --no-build-isolation \ + && BUILD_MONAI=1 FORCE_CUDA=1 pip install --no-cache-dir --no-build-isolation \ -r /tmp/rocm-requirements-$$.txt pytest -e . \ && rm -f /tmp/rocm-requirements-$$.txt +# Required at runtime too: monai.config.deviceconfig gates USE_COMPILED on this variable, so +# without it the extensions built above would be present but never used. +ENV BUILD_MONAI=1 + # Set HIPCIM_INDEX_URL="" to build without WSI/cucim support. # CuImage is imported (not just cucim) because cucim uses lazy_loader -- a bare import # succeeds even when the native library is unresolvable. Failing here is deliberate: if diff --git a/tests/networks/nets/test_densenet.py b/tests/networks/nets/test_densenet.py index 5521f461f7f..bc6d6d00948 100644 --- a/tests/networks/nets/test_densenet.py +++ b/tests/networks/nets/test_densenet.py @@ -95,8 +95,6 @@ def test_121_2d_shape_pretrain(self, model, input_param, input_shape, expected_s "ROCm may select different conv algorithms per graph; bit-exactness not guaranteed.", ) def test_pretrain_consistency(self, model, input_param, input_shape): - if torch.version.hip is not None: - self.skipTest("ROCm may select different conv algorithms per graph; bit-exactness not guaranteed.") example = torch.randn(input_shape).to(device) with skip_if_downloading_fails(): net = model(**input_param).to(device)