Skip to content

Commit b945eba

Browse files
committed
Merge current upstream dev into PR 9140 observation branch
Integrate upstream dev at 9a6ac14 without conflicts. Preserve the Python 3.12 dependency-resolution workflow byte for byte and retain every existing upstream gate. Validation: Ruby/Psych static YAML comparison, observational-job configuration assertions, bash -n for job run blocks, and git diff --cached --check. No Python, dependency installs, MONAI tests or Actions execution. Assisted-by: OpenAI Codex <noreply@openai.com> Signed-off-by: Demetrios Agourakis <demetrios@agourakis.med.br>
2 parents e970953 + 9a6ac14 commit b945eba

34 files changed

Lines changed: 1299 additions & 75 deletions

‎.pre-commit-config.yaml‎

Lines changed: 3 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -32,7 +32,7 @@ repos:
3232
- id: mixed-line-ending
3333

3434
- repo: https://github.com/astral-sh/ruff-pre-commit
35-
rev: v0.16.5
35+
rev: v0.16.10
3636
hooks:
3737
- id: ruff-check
3838
args: ["--fix"]
@@ -43,7 +43,7 @@ repos:
4343
)
4444
4545
- repo: https://github.com/psf/black-pre-commit-mirror
46-
rev: 26.5.1 # Black version, keep synced with MONAI requirements
46+
rev: 26.10.0 # Black version, keep synced with MONAI requirements
4747
hooks:
4848
- id: black
4949
language_version: python3
@@ -55,7 +55,7 @@ repos:
5555
)
5656
5757
- repo: https://github.com/pycqa/isort
58-
rev: 9.0.1 # isort version, keep synced with MONAI requirements
58+
rev: 9.0.2 # isort version, keep synced with MONAI requirements
5959
hooks:
6060
- id: isort
6161
name: isort (python)

‎Dockerfile.rocm‎

Lines changed: 126 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,126 @@
1+
# Copyright (c) MONAI Consortium
2+
# Licensed under the Apache License, Version 2.0 (the "License");
3+
# you may not use this file except in compliance with the License.
4+
# You may obtain a copy of the License at
5+
# http://www.apache.org/licenses/LICENSE-2.0
6+
# Unless required by applicable law or agreed to in writing, software
7+
# distributed under the License is distributed on an "AS IS" BASIS,
8+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
9+
# See the License for the specific language governing permissions and
10+
# limitations under the License.
11+
12+
# MONAI on AMD ROCm (AMD Instinct GPUs).
13+
#
14+
# This image installs the upstream `monai` package.
15+
#
16+
# docker build -f Dockerfile.rocm -t monai:rocm .
17+
#
18+
# docker run --device=/dev/kfd --device=/dev/dri --group-add video \
19+
# --ipc=host --shm-size=8g -it monai:rocm
20+
#
21+
# Select the GPU architecture with '--build-arg AMDGPU_TARGETS=gfx942|gfx950'. The default is gfx942.
22+
23+
ARG BASE_IMAGE=ubuntu:24.04
24+
FROM ${BASE_IMAGE}
25+
26+
LABEL maintainer="monai.contact@gmail.com"
27+
28+
ARG AMDGPU_TARGETS="gfx942"
29+
ARG ROCM_SERIES="10.0"
30+
ARG ROCM_INDEX_URL="https://stable.repo.amd.com/rocm/whl-next/"
31+
# Must match the python3 in BASE_IMAGE: it determines the venv's site-packages path below.
32+
ARG PYTHON_VERSION=3.12
33+
34+
ENV DEBIAN_FRONTEND=noninteractive
35+
36+
# ninja-build: required for MONAI's JIT C++/HIP extensions (torch.utils.cpp_extension).
37+
# libstdc++-13-dev + libopenslide-dev: HIP compile headers and whole-slide-image backend.
38+
RUN apt-get update \
39+
&& apt-get install -y --no-install-recommends \
40+
ca-certificates curl git openssh-client \
41+
python3-venv python3-pip python3-dev \
42+
build-essential cmake ninja-build yasm \
43+
libgomp1 libstdc++-13-dev \
44+
libopenslide-dev libwebp-dev libzstd-dev \
45+
&& rm -rf /var/lib/apt/lists/*
46+
47+
RUN python3 -m venv /opt/venv
48+
ENV PATH="/opt/venv/bin:${PATH}"
49+
RUN pip install --no-cache-dir --upgrade pip wheel
50+
51+
RUN pip install --no-cache-dir --index-url ${ROCM_INDEX_URL} \
52+
"rocm[libraries,devel,device-${AMDGPU_TARGETS}]==${ROCM_SERIES}.*" \
53+
"torch[device-${AMDGPU_TARGETS}]" \
54+
"torchvision[device-${AMDGPU_TARGETS}]" \
55+
torchaudio \
56+
&& rocm-sdk init
57+
58+
ENV ROCM_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_core"
59+
ENV ROCM_HOME="${ROCM_PATH}"
60+
ENV ROCM_DEVEL_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_devel"
61+
ENV ROCM_LIBRARIES_PATH="/opt/venv/lib/python${PYTHON_VERSION}/site-packages/_rocm_sdk_libraries"
62+
ENV PATH="${ROCM_PATH}/bin:${PATH}"
63+
ENV LD_LIBRARY_PATH="${ROCM_PATH}/lib:${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_LIBRARIES_PATH}/lib"
64+
ENV CPATH="${ROCM_DEVEL_PATH}/include:/usr/lib/gcc/x86_64-linux-gnu/13/include"
65+
ENV LIBRARY_PATH="${ROCM_DEVEL_PATH}/lib:${ROCM_PATH}/lib"
66+
ENV AMDGPU_TARGETS=${AMDGPU_TARGETS}
67+
ENV PYTORCH_ROCM_ARCH=${AMDGPU_TARGETS}
68+
69+
# hipcc expects bitcode at ROCM_PATH/amdgcn/bitcode; the rocm-sdk wheel places it one level deeper.
70+
RUN mkdir -p "${ROCM_PATH}/amdgcn" \
71+
&& ln -sf "${ROCM_PATH}/lib/llvm/amdgcn/bitcode" "${ROCM_PATH}/amdgcn/bitcode"
72+
73+
# Prevent OpenBLAS from spawning one thread per core under MONAI's multiprocessing dataloaders.
74+
ENV OMP_NUM_THREADS=1
75+
76+
WORKDIR /opt/monai
77+
78+
COPY LICENSE CHANGELOG.md CODE_OF_CONDUCT.md CONTRIBUTING.md README.md versioneer.py setup.py pyproject.toml runtests.sh MANIFEST.in ./
79+
COPY tests ./tests
80+
COPY monai ./monai
81+
82+
# Use print_dependencies.py rather than -e .[all,testing] to filter CUDA-only packages:
83+
# cucim-cu* pulls in cuda-toolkit (~1.2 GB); nvidia-ml-py fails at import on ROCm; nni depends on it.
84+
# BUILD_MONAI=1 builds the C++/HIP extensions ahead of time; FORCE_CUDA=1 is required because the
85+
# build host has no GPU, so setup.py's `torch.cuda.is_available()` check would otherwise skip them.
86+
# Compilation itself needs only the toolkit and PYTORCH_ROCM_ARCH (set above), not a device.
87+
# pytest is not in the "testing" extra; it is added explicitly for running tests by hand.
88+
RUN python monai/config/print_dependencies.py build-system \
89+
| xargs -d '\n' pip install --no-cache-dir --no-build-isolation \
90+
&& python monai/config/print_dependencies.py all testing \
91+
| grep -vE '^cucim-cu|^nvidia-ml-py|^nni' > /tmp/rocm-requirements-$$.txt \
92+
&& BUILD_MONAI=1 FORCE_CUDA=1 pip install --no-cache-dir --no-build-isolation \
93+
-r /tmp/rocm-requirements-$$.txt pytest -e . \
94+
&& rm -f /tmp/rocm-requirements-$$.txt
95+
96+
# Required at runtime too: monai.config.deviceconfig gates USE_COMPILED on this variable, so
97+
# without it the extensions built above would be present but never used.
98+
ENV BUILD_MONAI=1
99+
100+
# Set HIPCIM_INDEX_URL="" to build without WSI/cucim support.
101+
# CuImage is imported (not just cucim) because cucim uses lazy_loader -- a bare import
102+
# succeeds even when the native library is unresolvable. Failing here is deliberate: if
103+
# hipCIM was requested, an image where the cucim backends silently do not work is worse
104+
# than no image at all.
105+
ARG HIPCIM_INDEX_URL="https://pypi.amd.com/rocm-${ROCM_SERIES}.0/simple/"
106+
RUN if [ -n "${HIPCIM_INDEX_URL}" ]; then \
107+
pip install --no-cache-dir --extra-index-url "${HIPCIM_INDEX_URL}" "amd-hipcim" \
108+
&& python -c "from cucim import CuImage"; \
109+
else \
110+
echo "hipCIM not installed; whole-slide-image (cucim) backends are unavailable."; \
111+
fi
112+
113+
RUN python - <<'PY'
114+
import torch, monai
115+
print('MONAI :', monai.__version__)
116+
print('PyTorch:', torch.__version__)
117+
print('ROCm :', torch.version.hip)
118+
try:
119+
import cucim
120+
from cucim import CuImage # noqa: F401
121+
print('hipCIM :', cucim.__version__)
122+
except ImportError as exc:
123+
print('hipCIM : not available (%s)' % exc)
124+
PY
125+
126+
CMD ["bash"]

‎monai/_extensions/loader.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -75,7 +75,9 @@ def load_module(
7575
source = glob(path.join(module_dir, "**", "*.cpp"), recursive=True)
7676
if torch.cuda.is_available():
7777
source += glob(path.join(module_dir, "**", "*.cu"), recursive=True)
78-
platform_str += f"_{torch.version.cuda}"
78+
# `torch.version.cuda` is None on a ROCm build, which would make every ROCm
79+
# toolkit version share a single cache entry. Key on whichever is populated.
80+
platform_str += f"_{torch.version.cuda or f'hip{torch.version.hip}'}"
7981

8082
# Constructing compilation argument list.
8183
define_args = [] if not defines else [f"-D {key}={defines[key]}" for key in defines]

‎monai/apps/auto3dseg/bundle_gen.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -491,7 +491,7 @@ class BundleGen(AlgoGen):
491491
mlflow_tracking_uri: a tracking URI for MLflow server which could be local directory or address of
492492
the remote tracking Server; MLflow runs will be recorded locally in algorithms' model folder if
493493
the value is None.
494-
mlfow_experiment_name: a string to specify the experiment name for MLflow server.
494+
mlflow_experiment_name: a string to specify the experiment name for MLflow server.
495495
.. code-block:: bash
496496
497497
python -m monai.apps.auto3dseg BundleGen generate --data_stats_filename="../algorithms/datastats.yaml"

‎monai/apps/detection/metrics/matching.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -193,7 +193,7 @@ def _matching_no_gt(
193193
194194
Args:
195195
iou_thresholds: defined which IoU thresholds should be evaluated
196-
dt_scores: predicted scores
196+
pred_scores: predicted scores
197197
max_detections: maximum number of allowed detections per image.
198198
This functions uses this parameter to stay consistent with
199199
the actual matching function which needs this limit.

‎monai/data/dataset_summary.py‎

Lines changed: 16 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -159,13 +159,20 @@ def calculate_statistics(self, foreground_threshold: int = 0):
159159
label, *_ = convert_data_type(data=label, output_type=torch.Tensor)
160160

161161
image_foreground = image[torch.where(label > foreground_threshold)]
162+
if image_foreground.numel() == 0:
163+
continue
162164

163165
voxel_max.append(image_foreground.max().item())
164166
voxel_min.append(image_foreground.min().item())
165167
voxel_ct += len(image_foreground)
166168
voxel_sum += image_foreground.sum()
167169
voxel_square_sum += torch.square(image_foreground).sum()
168170

171+
if voxel_ct == 0:
172+
raise ValueError(
173+
f"No foreground voxels found in any sample with {foreground_threshold=}; "
174+
"set foreground_threshold=-1 to compute statistics over whole images."
175+
)
169176
self.data_max, self.data_min = max(voxel_max), min(voxel_min)
170177
self.data_mean = (voxel_sum / voxel_ct).item()
171178
self.data_std = (torch.sqrt(voxel_square_sum / voxel_ct - self.data_mean**2)).item()
@@ -204,11 +211,17 @@ def calculate_percentiles(
204211
label, *_ = convert_data_type(data=label, output_type=torch.Tensor)
205212

206213
intensities = image[torch.where(label > foreground_threshold)].tolist()
207-
if sampling_flag:
208-
intensities = intensities[::interval]
209-
all_intensities.append(intensities)
214+
if intensities:
215+
if sampling_flag:
216+
intensities = intensities[::interval]
217+
all_intensities.append(intensities)
210218

211219
all_intensities = list(chain(*all_intensities))
220+
if not all_intensities:
221+
raise ValueError(
222+
f"No foreground voxels found in any sample with {foreground_threshold=}; "
223+
"set foreground_threshold=-1 to compute statistics over whole images."
224+
)
212225
self.data_min_percentile, self.data_max_percentile = np.percentile(
213226
all_intensities, [min_percentile, max_percentile]
214227
)

‎monai/data/image_writer.py‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -266,12 +266,13 @@ def resample_if_needed(
266266
The output data type of this method is always ``np.float32``.
267267
"""
268268
orig_type = type(data_array)
269-
data_array = convert_to_tensor(data_array, track_meta=True)
269+
# Add the channel dimension before setting the affine so spatial_ndim is inferred correctly.
270+
data_array = convert_to_tensor(data_array, track_meta=True)[None]
270271
if affine is not None:
271272
data_array.affine = convert_to_tensor(affine, track_meta=False) # type: ignore
272273
resampler = SpatialResample(mode=mode, padding_mode=padding_mode, align_corners=align_corners, dtype=dtype)
273274
output_array = resampler(
274-
data_array[None], dst_affine=target_affine, spatial_size=output_spatial_shape # type: ignore
275+
data_array, dst_affine=target_affine, spatial_size=output_spatial_shape # type: ignore
275276
)
276277
# convert back at the end
277278
if isinstance(output_array, MetaTensor):

‎monai/data/utils.py‎

Lines changed: 17 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -686,28 +686,40 @@ def worker_init_fn(worker_id: int) -> None:
686686
set_rnd(worker_info.dataset, seed=worker_info.seed) # type: ignore[union-attr]
687687

688688

689-
def set_rnd(obj, seed: int) -> int:
689+
def set_rnd(obj, seed: int, _seen: set[int] | None = None) -> int:
690690
"""
691691
Set seed or random state for all randomizable properties of obj.
692692
693693
Args:
694694
obj: object to set seed or random state for.
695695
seed: set the random state with an integer seed.
696+
_seen: internal set of already-visited object ids, used to guard against
697+
infinite recursion on cyclic object graphs (e.g. OmegaConf/Hydra
698+
configs whose child nodes back-reference their parent, see issue #8087).
696699
"""
700+
if _seen is None:
701+
_seen = set()
697702
if isinstance(obj, (tuple, list)): # ZipDataset.data is a list
698-
_seed = seed
703+
if id(obj) in _seen:
704+
return seed
705+
_seen.add(id(obj))
706+
has_randomizable = False
699707
for item in obj:
700-
_seed = set_rnd(item, seed=seed)
701-
return seed if _seed == seed else seed + 1 # return a different seed if there are randomizable items
708+
item_seed = set_rnd(item, seed=seed, _seen=_seen)
709+
has_randomizable = has_randomizable or item_seed != seed
710+
return seed + 1 if has_randomizable else seed
702711
if not hasattr(obj, "__dict__"):
703712
return seed # no attribute
713+
if id(obj) in _seen:
714+
return seed # already visited: avoid infinite recursion on cyclic references
715+
_seen.add(id(obj))
704716
if hasattr(obj, "set_random_state"):
705717
obj.set_random_state(seed=seed % MAX_SEED)
706718
return seed + 1 # a different seed for the next component
707719
for key in obj.__dict__:
708720
if key.startswith("__"): # skip the private methods
709721
continue
710-
seed = set_rnd(obj.__dict__[key], seed=seed)
722+
seed = set_rnd(obj.__dict__[key], seed=seed, _seen=_seen)
711723
return seed
712724

713725

0 commit comments

Comments
 (0)