Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
126 changes: 126 additions & 0 deletions Dockerfile.rocm
Original file line number Diff line number Diff line change
@@ -0,0 +1,126 @@
# Copyright (c) MONAI Consortium
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
# http://www.apache.org/licenses/LICENSE-2.0
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

# MONAI on AMD ROCm (AMD Instinct GPUs).
#
# This image installs the upstream `monai` package.
#
# docker build -f Dockerfile.rocm -t monai:rocm .
#
# docker run --device=/dev/kfd --device=/dev/dri --group-add video \
# --ipc=host --shm-size=8g -it monai:rocm
#
# Select the GPU architecture with '--build-arg AMDGPU_TARGETS=gfx942|gfx950'. The default is gfx942.

ARG BASE_IMAGE=ubuntu:24.04
FROM ${BASE_IMAGE}

LABEL maintainer="monai.contact@gmail.com"

ARG AMDGPU_TARGETS="gfx942"
ARG ROCM_SERIES="10.0"
ARG ROCM_INDEX_URL="https://stable.repo.amd.com/rocm/whl-next/"
# Must match the python3 in BASE_IMAGE: it determines the venv's site-packages path below.
ARG PYTHON_VERSION=3.12

ENV DEBIAN_FRONTEND=noninteractive

# ninja-build: required for MONAI's JIT C++/HIP extensions (torch.utils.cpp_extension).
# libstdc++-13-dev + libopenslide-dev: HIP compile headers and whole-slide-image backend.
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
ca-certificates curl git openssh-client \
python3-venv python3-pip python3-dev \
build-essential cmake ninja-build yasm \
libgomp1 libstdc++-13-dev \
libopenslide-dev libwebp-dev libzstd-dev \
&& rm -rf /var/lib/apt/lists/*

RUN python3 -m venv /opt/venv
ENV PATH="/opt/venv/bin:${PATH}"
RUN pip install --no-cache-dir --upgrade pip wheel

RUN pip install --no-cache-dir --index-url ${ROCM_INDEX_URL} \
"rocm[libraries,devel,device-${AMDGPU_TARGETS}]==${ROCM_SERIES}.*" \
"torch[device-${AMDGPU_TARGETS}]" \
"torchvision[device-${AMDGPU_TARGETS}]" \
torchaudio \
&& rocm-sdk init

ENV ROCM_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_core"
ENV ROCM_HOME="${ROCM_PATH}"
ENV ROCM_DEVEL_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_devel"
ENV ROCM_LIBRARIES_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_libraries"
ENV PATH="${ROCM_PATH}/bin:${PATH}"
ENV LD_LIBRARY_PATH="${ROCM_PATH}/lib:${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_LIBRARIES_PATH}/lib"
ENV CPATH="${ROCM_DEVEL_PATH}/include:/usr/lib/gcc/x86_64-linux-gnu/13/include"
ENV LIBRARY_PATH="${ROCM_DEVEL_PATH}/lib:${ROCM_PATH}/lib"
ENV AMDGPU_TARGETS=${AMDGPU_TARGETS}
ENV PYTORCH_ROCM_ARCH=${AMDGPU_TARGETS}

# hipcc expects bitcode at ROCM_PATH/amdgcn/bitcode; the rocm-sdk wheel places it one level deeper.
RUN mkdir -p "${ROCM_PATH}/amdgcn" \
&& ln -sf "${ROCM_PATH}/lib/llvm/amdgcn/bitcode" "${ROCM_PATH}/amdgcn/bitcode"

# Prevent OpenBLAS from spawning one thread per core under MONAI's multiprocessing dataloaders.
ENV OMP_NUM_THREADS=1

WORKDIR /opt/monai

COPY LICENSE CHANGELOG.md CODE_OF_CONDUCT.md CONTRIBUTING.md README.md versioneer.py setup.py pyproject.toml runtests.sh MANIFEST.in ./
COPY tests ./tests
COPY monai ./monai

# Use print_dependencies.py rather than -e .[all,testing] to filter CUDA-only packages:
# cucim-cu* pulls in cuda-toolkit (~1.2 GB); nvidia-ml-py fails at import on ROCm; nni depends on it.
# BUILD_MONAI=1 builds the C++/HIP extensions ahead of time; FORCE_CUDA=1 is required because the
# build host has no GPU, so setup.py's `torch.cuda.is_available()` check would otherwise skip them.
# Compilation itself needs only the toolkit and PYTORCH_ROCM_ARCH (set above), not a device.
# pytest is not in the "testing" extra; it is added explicitly for running tests by hand.
RUN python monai/config/print_dependencies.py build-system \
| xargs -d '\n' pip install --no-cache-dir --no-build-isolation \
&& python monai/config/print_dependencies.py all testing \
| grep -vE '^cucim-cu|^nvidia-ml-py|^nni' > /tmp/rocm-requirements-$$.txt \
&& BUILD_MONAI=1 FORCE_CUDA=1 pip install --no-cache-dir --no-build-isolation \
-r /tmp/rocm-requirements-$$.txt pytest -e . \
&& rm -f /tmp/rocm-requirements-$$.txt

# Required at runtime too: monai.config.deviceconfig gates USE_COMPILED on this variable, so
# without it the extensions built above would be present but never used.
ENV BUILD_MONAI=1

# Set HIPCIM_INDEX_URL="" to build without WSI/cucim support.
# CuImage is imported (not just cucim) because cucim uses lazy_loader -- a bare import
# succeeds even when the native library is unresolvable. Failing here is deliberate: if
# hipCIM was requested, an image where the cucim backends silently do not work is worse
# than no image at all.
ARG HIPCIM_INDEX_URL="https://pypi.amd.com/rocm-${ROCM_SERIES}.0/simple/"
RUN if [ -n "${HIPCIM_INDEX_URL}" ]; then \
pip install --no-cache-dir --extra-index-url "${HIPCIM_INDEX_URL}" "amd-hipcim" \
&& python -c "from cucim import CuImage"; \
else \
echo "hipCIM not installed; whole-slide-image (cucim) backends are unavailable."; \
fi

RUN python - <<'PY'
import torch, monai
print('MONAI :', monai.__version__)
print('PyTorch:', torch.__version__)
print('ROCm :', torch.version.hip)
try:
import cucim
from cucim import CuImage # noqa: F401
print('hipCIM :', cucim.__version__)
except ImportError as exc:
print('hipCIM : not available (%s)' % exc)
PY

CMD ["bash"]
4 changes: 3 additions & 1 deletion monai/_extensions/loader.py
Original file line number Diff line number Diff line change
Expand Up @@ -75,7 +75,9 @@ def load_module(
source = glob(path.join(module_dir, "**", "*.cpp"), recursive=True)
if torch.cuda.is_available():
source += glob(path.join(module_dir, "**", "*.cu"), recursive=True)
platform_str += f"_{torch.version.cuda}"
# `torch.version.cuda` is None on a ROCm build, which would make every ROCm
# toolkit version share a single cache entry. Key on whichever is populated.
platform_str += f"_{torch.version.cuda or f'hip{torch.version.hip}'}"

# Constructing compilation argument list.
define_args = [] if not defines else [f"-D {key}={defines[key]}" for key in defines]
Expand Down
6 changes: 5 additions & 1 deletion setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,11 @@
BUILD_CPP = True
from torch.utils.cpp_extension import CUDA_HOME, CUDAExtension

BUILD_CUDA = FORCE_CUDA or (torch.cuda.is_available() and (CUDA_HOME is not None))
# On a ROCm build of torch, `CUDA_HOME` is None and the toolkit is located by `ROCM_HOME`
# instead; `CUDAExtension` hipifies the .cu sources transparently in that case. Accept
# either so the extensions are not silently skipped on ROCm.
_toolkit_home = CUDA_HOME or getattr(torch.utils.cpp_extension, "ROCM_HOME", None)
BUILD_CUDA = FORCE_CUDA or (torch.cuda.is_available() and (_toolkit_home is not None))

_pt_version = version.parse(torch.__version__).release
if _pt_version is None or len(_pt_version) < 3:
Expand Down
6 changes: 5 additions & 1 deletion tests/networks/nets/test_densenet.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@

import unittest
from typing import TYPE_CHECKING
from unittest import skipUnless
from unittest import skipIf, skipUnless

import torch
from parameterized import parameterized
Expand Down Expand Up @@ -90,6 +90,10 @@ def test_121_2d_shape_pretrain(self, model, input_param, input_shape, expected_s

@parameterized.expand([TEST_PRETRAINED_2D_CASE_3])
@skipUnless(has_torchvision, "Requires `torchvision` package.")
@skipIf(
torch.version.hip is not None,
"ROCm may select different conv algorithms per graph; bit-exactness not guaranteed.",
)
def test_pretrain_consistency(self, model, input_param, input_shape):
example = torch.randn(input_shape).to(device)
with skip_if_downloading_fails():
Expand Down
Loading