diff --git a/container-images/ramalama/Containerfile b/container-images/ramalama/Containerfile index d95c24d02..6fb4f9f70 100644 --- a/container-images/ramalama/Containerfile +++ b/container-images/ramalama/Containerfile @@ -8,6 +8,15 @@ RUN ./build_llama.sh ramalama FROM quay.io/fedora/fedora:44 +# NVIDIA_DRIVER_CAPABILITIES gates which driver libraries the NVIDIA container +# toolkit injects. The Vulkan ICD only comes in under the "graphics" capability, +# and the legacy nvidia-container-runtime hook that backs "docker --gpus all" +# defaults to "compute,utility" when the variable is unset, so the ICD is +# silently absent under Docker. The hook reads the variable from the image, so +# setting it here covers "ramalama run" and the generated quadlet, kube, compose +# and stack artifacts alike. CDI, which the podman path uses, ignores it. +ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility,graphics + RUN --mount=type=bind,from=builder,source=/tmp/install,target=/tmp/install,relabel=private \ cp -a /tmp/install/bin/ /usr/ && \ cp -a /tmp/install/lib64/*.so* /usr/lib64/ diff --git a/container-images/scripts/build_llama.sh b/container-images/scripts/build_llama.sh index 72cacb455..5adaddf2c 100755 --- a/container-images/scripts/build_llama.sh +++ b/container-images/scripts/build_llama.sh @@ -142,6 +142,11 @@ dnf_install_runtime_deps() { if [ "$uname_m" = "x86_64" ] || [ "$uname_m" = "aarch64" ]; then dnf copr enable -y slp/mesa-libkrun-vulkan runtime_pkgs+=(vulkan-loader vulkan-tools "mesa-vulkan-drivers-$MESA_VULKAN_VERSION") + # NVIDIA's Vulkan ICD (libGLX_nvidia.so.0, injected by the container + # toolkit) needs libXext and libEGL.so.1 present to initialize. + # libglvnd-egl pulls in mesa-libEGL, so pin it to the copr version to + # avoid dragging the rest of mesa off MESA_VULKAN_VERSION. + runtime_pkgs+=(libXext libglvnd-egl "mesa-libEGL-$MESA_VULKAN_VERSION") else runtime_pkgs+=(openblas) fi diff --git a/docs/options/backend.md b/docs/options/backend.md index 2b9c9d70b..41d6a40da 100644 --- a/docs/options/backend.md +++ b/docs/options/backend.md @@ -10,7 +10,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -21,7 +21,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-bench.1.md b/docs/ramalama-bench.1.md index 6b77633e7..f82372170 100644 --- a/docs/ramalama-bench.1.md +++ b/docs/ramalama-bench.1.md @@ -46,7 +46,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -57,7 +57,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-perplexity.1.md b/docs/ramalama-perplexity.1.md index bb6833404..6cff430ba 100644 --- a/docs/ramalama-perplexity.1.md +++ b/docs/ramalama-perplexity.1.md @@ -46,7 +46,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -57,7 +57,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-run.1.md b/docs/ramalama-run.1.md index 059c97e4b..5ed45f709 100644 --- a/docs/ramalama-run.1.md +++ b/docs/ramalama-run.1.md @@ -58,7 +58,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -69,7 +69,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-sandbox-goose.1.md b/docs/ramalama-sandbox-goose.1.md index 85c232d64..35c1fd3af 100644 --- a/docs/ramalama-sandbox-goose.1.md +++ b/docs/ramalama-sandbox-goose.1.md @@ -54,7 +54,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -65,7 +65,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-sandbox-opencode.1.md b/docs/ramalama-sandbox-opencode.1.md index 379b1c843..5af6bf194 100644 --- a/docs/ramalama-sandbox-opencode.1.md +++ b/docs/ramalama-sandbox-opencode.1.md @@ -54,7 +54,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -65,7 +65,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-sandbox-pi.1.md b/docs/ramalama-sandbox-pi.1.md index 396508d20..868af0d3e 100644 --- a/docs/ramalama-sandbox-pi.1.md +++ b/docs/ramalama-sandbox-pi.1.md @@ -58,7 +58,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -69,7 +69,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-serve.1.md b/docs/ramalama-serve.1.md index 1c604a1c0..83400ba02 100644 --- a/docs/ramalama-serve.1.md +++ b/docs/ramalama-serve.1.md @@ -87,7 +87,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -98,7 +98,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama.conf b/docs/ramalama.conf index 4390b3b45..cc54ea239 100644 --- a/docs/ramalama.conf +++ b/docs/ramalama.conf @@ -241,12 +241,12 @@ # Valid options: auto, vulkan, rocm, cuda, sycl, openvino, cann, musa # - auto (default): Automatically selects the preferred backend based on detected GPU # - AMD GPUs: vulkan (Linux/macOS) or rocm (Windows) -# - NVIDIA GPUs: cuda +# - NVIDIA GPUs: cuda; vulkan available as explicit option # - Intel GPUs: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option # - Ascend NPUs: cann # - MUSA GPUs: musa # - No GPU: vulkan (CPU fallback) -# - vulkan: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +# - vulkan: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) # - rocm: Use AMD ROCm backend (AMD GPUs only) # - cuda: Use NVIDIA CUDA backend (NVIDIA GPUs only) # - sycl: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama.conf.5.md b/docs/ramalama.conf.5.md index 641304153..529913a09 100644 --- a/docs/ramalama.conf.5.md +++ b/docs/ramalama.conf.5.md @@ -240,13 +240,13 @@ Valid options: `auto`, `vulkan`, `rocm`, `cuda`, `sycl`, `openvino`, `cann`, `mu - **auto** (default): Automatically selects the preferred backend based on detected GPU: - AMD GPUs: vulkan (Linux/macOS) or rocm (Windows) - - NVIDIA GPUs: cuda + - NVIDIA GPUs: cuda; vulkan available as explicit option - Intel GPUs: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - Ascend NPUs: cann - MUSA GPUs: musa - No GPU: vulkan (CPU fallback) -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/ramalama/cli.py b/ramalama/cli.py index 1758f30d3..2a149c6f3 100644 --- a/ramalama/cli.py +++ b/ramalama/cli.py @@ -125,6 +125,15 @@ def parse_port_option(option: str) -> str: class OverrideDefaultAction(argparse.Action): + # Options whose default is computed while the parser is built, before the + # rest of the command line has been seen. Recorded so such a default can be + # told apart from a value the user passed. + dests: set[str] = set() + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + OverrideDefaultAction.dests.add(self.dest) + def __call__(self, parser, namespace, values, option_string=None): setattr(namespace, self.dest, values) setattr(namespace, self.dest + '_override', True) @@ -221,11 +230,29 @@ def parse_args_from_cmd(cmd: list[str]) -> tuple[argparse.ArgumentParser, argpar post_parse_setup(args) for arg in args.__dict__.keys() & config._fields: + if arg in OverrideDefaultAction.dests: + # These options record whether they were passed, so go by that rather + # than by the value. A computed default is not the user's choice and + # does not belong in the config: --image defaults to the image the + # detected GPU resolves to, worked out before --backend was parsed, + # and storing it marks the image as chosen rather than derived. A + # value the user did pass belongs there even where it matches the + # default, which is how "--image $RAMALAMA_DEFAULT_IMAGE" asks for + # that image instead of the one the GPU would select. + if getattr(args, f"{arg}_override", False): + setattr(config, arg, getattr(args, arg)) + continue if getattr(args, arg) != getattr(config, arg): setattr(config, arg, getattr(args, arg)) runtime_plugin = get_runtime(config.runtime) runtime_plugin.sync_args_to_runtime_config(args, config) + + # The runtime config now carries --backend, so the image the command will + # run can be resolved for real. + if hasattr(args, "image") and not getattr(args, "image_override", False): + args.image = accel_image(config) + return parser, args diff --git a/ramalama/common.py b/ramalama/common.py index 2987d6cb1..fa38f08bf 100644 --- a/ramalama/common.py +++ b/ramalama/common.py @@ -57,6 +57,22 @@ def sanitize_filename(filename: str) -> str: podman_machine_accel = False +# CDI device names of the NVIDIA GPUs the user narrowed CUDA_VISIBLE_DEVICES +# down to, set by check_nvidia(). Empty when every detected GPU is in play. +nvidia_selected_devices: list[str] = [] + + +def container_cuda_visible_devices(value: str) -> str: + """CUDA_VISIBLE_DEVICES as it should read inside the container. + + Where the selection is narrowed, only those GPUs are passed in and the + container numbers them from zero, so the host's indices no longer describe + them. Returns the value unchanged when every GPU is in play. + """ + if not nvidia_selected_devices: + return value + return ",".join(str(i) for i in range(len(nvidia_selected_devices))) + def confirm_no_gpu(name, provider) -> bool: while True: @@ -438,9 +454,13 @@ def find_in_cdi(devices: list[str]) -> tuple[list[str], list[str]]: for device in devices: if device in cdi_device_names: configured.append(device) - # A device can be specified by a prefix of the uuid - elif device.startswith("GPU") and any(name.startswith(device) for name in cdi_device_names): - configured.append(device) + # A device can be specified by a prefix of the uuid. Record the full + # name it resolves to: it reaches "--device nvidia.com/gpu=", + # which matches the CDI configuration exactly and not by prefix. + elif device.startswith("GPU") and ( + full_name := next((name for name in cdi_device_names if name.startswith(device)), None) + ): + configured.append(full_name) else: perror(f"Device {device} does not have a CDI configuration") unconfigured.append(device) @@ -468,6 +488,20 @@ def check_metal(args: ContainerArgType) -> bool: return platform.system() == "Darwin" +@lru_cache(maxsize=1) +def has_nvidia_vulkan_icd() -> bool: + """True when NVIDIA's Vulkan ICD manifest is installed on the host. + + The container toolkit injects the ICD from the host driver installation, so + when it is missing here the vulkan backend finds only mesa's llvmpipe inside + the container and runs on the CPU instead of failing. + """ + return any( + glob.glob(os.path.join(host_path(icd_dir), "*nvidia*.json")) + for icd_dir in ("/usr/share/vulkan/icd.d", "/etc/vulkan/icd.d") + ) + + @lru_cache(maxsize=1) def check_nvidia() -> Optional[Literal["cuda"]]: try: @@ -515,6 +549,16 @@ def check_nvidia() -> Optional[Literal["cuda"]]: if not configured: configured = indices + # Record a narrowed selection so the engine can pass just those GPUs + # into the container. Not every backend filters on CUDA_VISIBLE_DEVICES + # (llama.cpp's Vulkan backend indexes Vulkan's own device enumeration), + # so the selection has to happen at the device level to be honoured. + # These names came back from find_in_cdi(), so they are known to the CDI + # configuration; the fallback above to every index has not been checked + # and stays on the "all" device. + global nvidia_selected_devices + nvidia_selected_devices = configured if set(configured) != set(indices) else [] + os.environ["CUDA_VISIBLE_DEVICES"] = ','.join(configured) return "cuda" diff --git a/ramalama/compose.py b/ramalama/compose.py index 1c39000de..ab53e62c5 100644 --- a/ramalama/compose.py +++ b/ramalama/compose.py @@ -5,7 +5,9 @@ import shlex from typing import Optional -from ramalama.common import RAG_DIR, get_accel_env_vars, get_gpu_devices +# Live reference for checking global vars +import ramalama.common +from ramalama.common import RAG_DIR, container_cuda_visible_devices, get_accel_env_vars, get_gpu_devices from ramalama.file import PlainFile from ramalama.host_utils import format_bind_host_publish_prefix from ramalama.version import version @@ -122,6 +124,10 @@ def _gen_ports(self) -> str: def _gen_environment(self) -> str: env_vars = get_accel_env_vars() + if "CUDA_VISIBLE_DEVICES" in env_vars: + # device_ids below reserves just the selected GPUs, so the container + # renumbers them, exactly as it does for "ramalama run". + env_vars["CUDA_VISIBLE_DEVICES"] = container_cuda_visible_devices(env_vars["CUDA_VISIBLE_DEVICES"]) # Allow user to override with --env if getattr(self.args, "env", None): for e in self.args.env: @@ -137,17 +143,29 @@ def _gen_environment(self) -> str: return env_spec def _gen_gpu_deployment(self) -> str: - gpu_keywords = ["cuda", "rocm", "gpu"] - if not any(keyword in self.image.lower() for keyword in gpu_keywords): + # The "nvidia" device driver only covers NVIDIA GPUs, so key the + # reservation off the detected hardware. The image name does not + # identify it: NVIDIA can be served by the vulkan-capable ramalama + # image as well as by the cuda one. + if "CUDA_VISIBLE_DEVICES" not in get_accel_env_vars(): return "" - return """\ + # Reserve only the GPUs the user selected. CUDA_VISIBLE_DEVICES alone + # cannot narrow it: the Vulkan backend never reads that variable, so the + # other GPUs have to be kept out of the container entirely. + if selected := ramalama.common.nvidia_selected_devices: + device_ids = ", ".join(f'"{name}"' for name in selected) + reservation = f"device_ids: [{device_ids}]" + else: + reservation = "count: all" + + return f"""\ deploy: resources: reservations: devices: - driver: nvidia - count: all + {reservation} capabilities: [gpu]""" def _gen_command(self) -> str: diff --git a/ramalama/engine.py b/ramalama/engine.py index 90aa125eb..01ffdadc1 100644 --- a/ramalama/engine.py +++ b/ramalama/engine.py @@ -15,7 +15,17 @@ # Live reference for checking global vars import ramalama.common from ramalama.arg_types import BaseEngineArgsType -from ramalama.common import check_nvidia, engine_cmd, exec_cmd, genname, get_accel_env_vars, host_path, perror, run_cmd +from ramalama.common import ( + check_nvidia, + container_cuda_visible_devices, + engine_cmd, + exec_cmd, + genname, + get_accel_env_vars, + host_path, + perror, + run_cmd, +) from ramalama.compat import NamedTemporaryFile from ramalama.config import ActiveConfig from ramalama.host_utils import ( @@ -118,11 +128,23 @@ def add_device_options(self): for k, v in get_accel_env_vars().items(): # Special case for Cuda if k == "CUDA_VISIBLE_DEVICES": + # Pass in only the GPUs the user selected rather than all of + # them plus CUDA_VISIBLE_DEVICES to filter with: the Vulkan + # backend never reads that variable, it indexes Vulkan's own + # device enumeration, so a narrowed selection is only honoured + # if the container cannot see the other GPUs in the first place. + selected = ramalama.common.nvidia_selected_devices if self.use_docker: - self.exec_args += ["--gpus", "all"] + # docker's csv parser needs the quotes to keep a + # comma-separated device list in one field. + self.exec_args += ["--gpus", f'"device={",".join(selected)}"' if selected else "all"] + elif selected: + for name in selected: + self.exec_args += ["--device", f"nvidia.com/gpu={name}"] else: # newer Podman versions support --gpus=all, but < 5.0 do not self.exec_args += ["--device", "nvidia.com/gpu=all"] + v = container_cuda_visible_devices(v) elif k == "MUSA_VISIBLE_DEVICES": self.exec_args += ["--env", "MTHREADS_VISIBLE_DEVICES=all"] elif k == "INTEL_VISIBLE_DEVICES": diff --git a/ramalama/plugins/runtimes/inference/llama_cpp.py b/ramalama/plugins/runtimes/inference/llama_cpp.py index 97a8fefe5..0a4a537e6 100644 --- a/ramalama/plugins/runtimes/inference/llama_cpp.py +++ b/ramalama/plugins/runtimes/inference/llama_cpp.py @@ -12,6 +12,7 @@ from collections.abc import Callable, Mapping from dataclasses import asdict, dataclass, field from datetime import datetime, timezone +from functools import lru_cache from http.client import HTTPConnection from typing import Any, Literal, Optional, get_args from urllib.parse import urlparse @@ -40,6 +41,7 @@ ensure_image, genname, get_gpu_type_env_vars, + has_nvidia_vulkan_icd, run_cmd, set_accel_env_vars, set_gpu_type_env_vars, @@ -144,7 +146,7 @@ def get_gpu_backend_preferences(gpu_type: str) -> list[str]: preferences = { "HIP_VISIBLE_DEVICES": ["vulkan", "rocm"], # AMD: Vulkan preferred - "CUDA_VISIBLE_DEVICES": ["cuda"], # NVIDIA: CUDA only + "CUDA_VISIBLE_DEVICES": ["cuda", "vulkan"], # NVIDIA: CUDA preferred "INTEL_VISIBLE_DEVICES": ["vulkan", "sycl", "openvino"], # Intel: Vulkan preferred "ASAHI_VISIBLE_DEVICES": ["vulkan"], # Asahi: Vulkan only "ASCEND_VISIBLE_DEVICES": ["cann"], # Ascend: CANN only @@ -159,6 +161,22 @@ def get_gpu_backend_preferences(gpu_type: str) -> list[str]: return preferences.get(gpu_type, []) +@lru_cache(maxsize=1) +def warn_without_nvidia_vulkan_icd() -> None: + """Warn once when vulkan is picked for an NVIDIA GPU the host cannot drive. + + Vulkan is only usable on NVIDIA because the container toolkit injects the + vendor ICD. Without it llama.cpp silently serves from the CPU, which is far + slower but never fails, so say so up front.""" + if has_nvidia_vulkan_icd(): + return + logger.warning( + "No NVIDIA Vulkan ICD found in /usr/share/vulkan/icd.d or /etc/vulkan/icd.d. " + "Inference may fall back to the CPU. Install the nvidia-container-toolkit, which " + "provides the ICD for the container, or select the cuda backend with --backend cuda." + ) + + def backend_to_gpu_env(backend: str) -> str: """Maps a backend name to its corresponding GPU environment variable.""" mapping = { @@ -353,6 +371,9 @@ def get_container_image(self, config: Any, detected_gpu_type: str) -> Optional[s f"Recommended backends for {gpu_name}: {', '.join(preferences)}" ) + if gpu_type == "GGML_VK_VISIBLE_DEVICES" and detected_gpu_type == "CUDA_VISIBLE_DEVICES": + warn_without_nvidia_vulkan_icd() + override = config.images.get(gpu_type) if gpu_type else None return override if override else _LLAMA_CPP_IMAGES.get(gpu_type, config.default_image) diff --git a/test/unit/conftest.py b/test/unit/conftest.py index 3adaacb17..50d6bb225 100644 --- a/test/unit/conftest.py +++ b/test/unit/conftest.py @@ -45,6 +45,20 @@ def _isolate_from_toolbox(): in_toolbox.cache_clear() +@pytest.fixture(autouse=True) +def _clear_vulkan_icd_cache(): + """The ICD probe and the warning it drives are cached for the process, so + clear them around every test rather than carrying the host's answer.""" + from ramalama.common import has_nvidia_vulkan_icd + from ramalama.plugins.runtimes.inference.llama_cpp import warn_without_nvidia_vulkan_icd + + for cached in (has_nvidia_vulkan_icd, warn_without_nvidia_vulkan_icd): + cached.cache_clear() + yield + for cached in (has_nvidia_vulkan_icd, warn_without_nvidia_vulkan_icd): + cached.cache_clear() + + @pytest.fixture def force_oci_image(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr(OCIStrategyFactory, "resolve", lambda self, model: self.strategies("image")) diff --git a/test/unit/data/test_compose/with_amd_gpu.yaml b/test/unit/data/test_compose/with_amd_gpu.yaml new file mode 100644 index 000000000..e69080f6e --- /dev/null +++ b/test/unit/data/test_compose/with_amd_gpu.yaml @@ -0,0 +1,18 @@ +# Save this output to a 'docker-compose.yaml' file and run 'docker compose up'. +# +# Created with ramalama-0.1.0-test +services: + gemma-rocm: + container_name: ramalama-gemma-rocm + image: test-image/rocm:latest + volumes: + - "/models/gemma.gguf:/mnt/models/gemma.gguf:ro" + ports: + - "8080:8080" + environment: + - HIP_VISIBLE_DEVICES=0 + devices: + - "/dev/accel:/dev/accel" + - "/dev/dri:/dev/dri" + - "/dev/kfd:/dev/kfd" + restart: unless-stopped diff --git a/test/unit/data/test_compose/with_nvidia_gpu.yaml b/test/unit/data/test_compose/with_nvidia_gpu.yaml index 21f008cca..d0768ce29 100644 --- a/test/unit/data/test_compose/with_nvidia_gpu.yaml +++ b/test/unit/data/test_compose/with_nvidia_gpu.yaml @@ -10,7 +10,7 @@ services: ports: - "8080:8080" environment: - - ACCEL_ENV=true + - CUDA_VISIBLE_DEVICES=0 devices: - "/dev/accel:/dev/accel" - "/dev/dri:/dev/dri" diff --git a/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml b/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml new file mode 100644 index 000000000..e9905e863 --- /dev/null +++ b/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml @@ -0,0 +1,25 @@ +# Save this output to a 'docker-compose.yaml' file and run 'docker compose up'. +# +# Created with ramalama-0.1.0-test +services: + gemma-cuda: + container_name: ramalama-gemma-cuda + image: test-image/cuda:latest + volumes: + - "/models/gemma.gguf:/mnt/models/gemma.gguf:ro" + ports: + - "8080:8080" + environment: + - CUDA_VISIBLE_DEVICES=0 + devices: + - "/dev/accel:/dev/accel" + - "/dev/dri:/dev/dri" + - "/dev/kfd:/dev/kfd" + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["1"] + capabilities: [gpu] + restart: unless-stopped diff --git a/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml b/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml new file mode 100644 index 000000000..8c7aed896 --- /dev/null +++ b/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml @@ -0,0 +1,25 @@ +# Save this output to a 'docker-compose.yaml' file and run 'docker compose up'. +# +# Created with ramalama-0.1.0-test +services: + gemma-vulkan: + container_name: ramalama-gemma-vulkan + image: test-image/ramalama:latest + volumes: + - "/models/gemma.gguf:/mnt/models/gemma.gguf:ro" + ports: + - "8080:8080" + environment: + - CUDA_VISIBLE_DEVICES=0 + devices: + - "/dev/accel:/dev/accel" + - "/dev/dri:/dev/dri" + - "/dev/kfd:/dev/kfd" + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: all + capabilities: [gpu] + restart: unless-stopped diff --git a/test/unit/test_common.py b/test/unit/test_common.py index f914059c4..cd074e1b5 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -9,6 +9,7 @@ import pytest +import ramalama.common from ramalama.cli import ( default_image, default_rag_image, @@ -20,10 +21,12 @@ accel_image, check_intel, check_nvidia, + container_cuda_visible_devices, engine_cmd, ensure_image, find_in_cdi, get_accel, + has_nvidia_vulkan_icd, host_available, host_cmd, host_path, @@ -32,6 +35,7 @@ populate_volume_from_image, rm_until_substring, verify_checksum, + version_tagged_image, ) from ramalama.compat import NamedTemporaryFile from ramalama.config import DEFAULT_IMAGE, load_config @@ -137,6 +141,9 @@ def test_verify_checksum( ("HIP_VISIBLE_DEVICES", f"{_BASE_IMAGE}:latest", None, None, f"{_BASE_IMAGE}:latest"), ("HIP_VISIBLE_DEVICES", None, f"{_BASE_IMAGE}:latest", None, f"{_BASE_IMAGE}:latest"), ("HIP_VISIBLE_DEVICES", None, None, f"{_BASE_IMAGE}:latest", f"{_BASE_IMAGE}:latest"), + # An --image that happens to match the default is still the user asking + # for it, and wins over the image the detected GPU would select. + ("CUDA_VISIBLE_DEVICES", None, None, DEFAULT_IMAGE, DEFAULT_IMAGE), ], ) def test_accel_image( @@ -176,6 +183,35 @@ def test_accel_image( assert accel_image(config) == expected_result +@pytest.mark.parametrize( + "accel_env,backend,expected_result", + [ + # Left on auto, the detected GPU picks the image. + ("CUDA_VISIBLE_DEVICES", "auto", version_tagged_image("quay.io/ramalama/cuda")), + ("CUDA_VISIBLE_DEVICES", "cuda", version_tagged_image("quay.io/ramalama/cuda")), + # A backend the user asked for wins over the detected GPU, even though + # --image defaults to the image that GPU resolves to. + ("CUDA_VISIBLE_DEVICES", "vulkan", DEFAULT_IMAGE), + ("HIP_VISIBLE_DEVICES", "auto", DEFAULT_IMAGE), + ("HIP_VISIBLE_DEVICES", "vulkan", DEFAULT_IMAGE), + ("HIP_VISIBLE_DEVICES", "rocm", version_tagged_image("quay.io/ramalama/rocm")), + ], +) +def test_accel_image_follows_backend(accel_env: str, backend: str, expected_result: str, monkeypatch): + monkeypatch.setattr("ramalama.common.get_accel", lambda: "none") + + env = {"RAMALAMA_CONFIG": "/dev/null", accel_env: "1"} + with patch.dict("os.environ", env, clear=True): + config = load_config() + with patch("ramalama.cli.ActiveConfig", return_value=config): + default_image.cache_clear() + default_rag_image.cache_clear() + default_tools_image.cache_clear() + _, args = parse_args_from_cmd(["run", "--backend", backend, "granite"]) + assert accel_image(config) == expected_result + assert args.image == expected_result + + @patch("ramalama.common.run_cmd") @patch("ramalama.common.handle_provider") def test_apple_vm_returns_result_podman_v5(mock_handle_provider, mock_run_cmd): @@ -287,6 +323,14 @@ def test_pull_fails_ramalama_image_fallback_fails_raises(self, mock_run_cmd): class TestCheckNvidia: def setup_method(self): check_nvidia.cache_clear() + self.cuda_visible_devices = os.environ.pop("CUDA_VISIBLE_DEVICES", None) + ramalama.common.nvidia_selected_devices = [] + + def teardown_method(self): + os.environ.pop("CUDA_VISIBLE_DEVICES", None) + if self.cuda_visible_devices is not None: + os.environ["CUDA_VISIBLE_DEVICES"] = self.cuda_visible_devices + ramalama.common.nvidia_selected_devices = [] @patch("ramalama.common.find_in_cdi") @patch("ramalama.common.run_cmd") @@ -318,6 +362,51 @@ def test_check_nvidia_no_cdi_toolkit_present(self, mock_run_cmd, mock_find_in_cd printed = " ".join(str(c.args[0]) for c in mock_perror.call_args_list) assert "nvidia-ctk cdi generate" in printed + @patch("ramalama.common.find_in_cdi") + @patch("ramalama.common.run_cmd") + def test_check_nvidia_all_gpus_are_not_a_selection(self, mock_run_cmd, mock_find_in_cdi): + mock_find_in_cdi.return_value = (["all"], []) + mock_run_cmd.return_value.stdout = "0,GPU-1111\n1,GPU-2222" + assert check_nvidia() == "cuda" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "0,1" + assert ramalama.common.nvidia_selected_devices == [] + + @patch("ramalama.common.find_in_cdi") + @patch("ramalama.common.run_cmd") + def test_check_nvidia_records_narrowed_selection(self, mock_run_cmd, mock_find_in_cdi): + os.environ["CUDA_VISIBLE_DEVICES"] = "1" + mock_find_in_cdi.return_value = (["1", "all"], []) + mock_run_cmd.return_value.stdout = "0,GPU-1111\n1,GPU-2222" + assert check_nvidia() == "cuda" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1" + assert ramalama.common.nvidia_selected_devices == ["1"] + + @patch("ramalama.common.find_in_cdi") + @patch("ramalama.common.run_cmd") + def test_check_nvidia_selection_needs_a_cdi_device(self, mock_run_cmd, mock_find_in_cdi): + # Only the "all" device is configured, so the narrowing cannot be + # expressed as a device and every GPU stays visible, as before. + os.environ["CUDA_VISIBLE_DEVICES"] = "1" + mock_find_in_cdi.return_value = (["all"], ["1"]) + mock_run_cmd.return_value.stdout = "0,GPU-1111\n1,GPU-2222" + assert check_nvidia() == "cuda" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "0,1" + assert ramalama.common.nvidia_selected_devices == [] + + @pytest.mark.parametrize( + "selected,value,expected", + [ + # Every GPU is in play, so the host's numbering still describes them. + ([], "0,1", "0,1"), + (["1"], "1", "0"), + (["1", "2"], "1,2", "0,1"), + (["GPU-2222"], "GPU-2222", "0"), + ], + ) + def test_container_cuda_visible_devices(self, selected, value, expected): + with patch.object(ramalama.common, "nvidia_selected_devices", selected): + assert container_cuda_visible_devices(value) == expected + @patch("ramalama.common.run_cmd") def test_check_nvidia_smi_failure(self, mock_run_cmd): mock_run_cmd.side_effect = subprocess.CalledProcessError(1, "nvidia-smi") @@ -610,6 +699,9 @@ def open_side_effect(path, *args, **kwargs): (["all"], ["all"], []), (["0", "all"], ["0", "all"], []), ([CDI_GPU_UUID, "all"], [CDI_GPU_UUID, "all"], []), + # An abbreviated uuid resolves to the full CDI device name, which is + # what "--device nvidia.com/gpu=" needs. + ([CDI_GPU_UUID[:12], "all"], [CDI_GPU_UUID, "all"], []), (["1", "all"], ["all"], ["1"]), (["dummy", "all"], ["all"], ["dummy"]), ], @@ -858,3 +950,17 @@ def test_host_path_in_toolbox_no_run_host(self): patch("os.path.isdir", return_value=False), ): assert host_path("/etc/cdi") == "/etc/cdi" + + +class TestHasNvidiaVulkanIcd: + @pytest.mark.parametrize("icd_dir", ["/usr/share/vulkan/icd.d", "/etc/vulkan/icd.d"]) + def test_icd_installed(self, icd_dir): + with patch( + "ramalama.common.glob.glob", + side_effect=lambda p: [f"{icd_dir}/nvidia_icd.json"] if p.startswith(icd_dir) else [], + ): + assert has_nvidia_vulkan_icd() + + def test_only_other_vendors(self): + with patch("ramalama.common.glob.glob", return_value=[]): + assert not has_nvidia_vulkan_icd() diff --git a/test/unit/test_compose.py b/test/unit/test_compose.py index 42c32b7ef..694cad764 100644 --- a/test/unit/test_compose.py +++ b/test/unit/test_compose.py @@ -37,6 +37,8 @@ def __init__( mmproj_file_exists: bool = False, args: Args = Args(), exec_args: list = None, + accel_env_vars: dict = None, + nvidia_selected_devices: list = None, ): self.model_name = model_name self.model_src_path = model_src_path @@ -53,6 +55,8 @@ def __init__( self.mmproj_file_exists = mmproj_file_exists self.args = args self.exec_args = exec_args if exec_args is not None else [] + self.accel_env_vars = accel_env_vars if accel_env_vars is not None else {"ACCEL_ENV": "true"} + self.nvidia_selected_devices = nvidia_selected_devices or [] DATA_PATH = Path(__file__).parent / "data" / "test_compose" @@ -156,9 +160,49 @@ def __init__( model_src_path="/models/gemma.gguf", model_dest_path="/mnt/models/gemma.gguf", args=Args(image="test-image/cuda:latest"), + accel_env_vars={"CUDA_VISIBLE_DEVICES": "0"}, ), "with_nvidia_gpu.yaml", ), + ( + # The reservation follows the detected GPU, not the image name, so + # an NVIDIA GPU served by the vulkan-capable ramalama image gets one + # too. + Input( + model_name="gemma-vulkan", + model_src_path="/models/gemma.gguf", + model_dest_path="/mnt/models/gemma.gguf", + accel_env_vars={"CUDA_VISIBLE_DEVICES": "0"}, + ), + "with_nvidia_gpu_vulkan_image.yaml", + ), + ( + # A narrowed selection reserves just those GPUs, since the Vulkan + # backend cannot be filtered with CUDA_VISIBLE_DEVICES. The variable + # is renumbered to match what the container ends up seeing. + Input( + model_name="gemma-cuda", + model_src_path="/models/gemma.gguf", + model_dest_path="/mnt/models/gemma.gguf", + args=Args(image="test-image/cuda:latest"), + accel_env_vars={"CUDA_VISIBLE_DEVICES": "1"}, + nvidia_selected_devices=["1"], + ), + "with_nvidia_gpu_selection.yaml", + ), + ( + # The other direction: "nvidia" is the only driver the reservation + # can name, so an AMD GPU must not get one whatever the image is + # called. + Input( + model_name="gemma-rocm", + model_src_path="/models/gemma.gguf", + model_dest_path="/mnt/models/gemma.gguf", + args=Args(image="test-image/rocm:latest"), + accel_env_vars={"HIP_VISIBLE_DEVICES": "0"}, + ), + "with_amd_gpu.yaml", + ), ( Input( model_name="tinyllama", @@ -188,7 +232,8 @@ def test_compose_generate(input_data: Input, expected_file_name: str, monkeypatc } monkeypatch.setattr("os.path.exists", lambda path: existence.get(path, False)) - monkeypatch.setattr("ramalama.compose.get_accel_env_vars", lambda: {"ACCEL_ENV": "true"}) + monkeypatch.setattr("ramalama.compose.get_accel_env_vars", lambda: dict(input_data.accel_env_vars)) + monkeypatch.setattr("ramalama.common.nvidia_selected_devices", list(input_data.nvidia_selected_devices)) monkeypatch.setattr("ramalama.compose.version", lambda: "0.1.0-test") draft_model_paths = (None, None) diff --git a/test/unit/test_engine.py b/test/unit/test_engine.py index b86d16bc0..7f2b0c78e 100644 --- a/test/unit/test_engine.py +++ b/test/unit/test_engine.py @@ -27,6 +27,37 @@ def test_init_basic(self): self.assertEqual(engine.use_podman, True) self.assertEqual(engine.use_docker, False) + def _cuda_device_args(self, engine_name, visible_devices, selected=()): + args = Namespace(**{**vars(self.base_args), "engine": engine_name}) + engine = ramalama.engine.Engine(args) + engine.exec_args = [] + with ( + patch("ramalama.engine.get_accel_env_vars", return_value={"CUDA_VISIBLE_DEVICES": visible_devices}), + patch("glob.glob", return_value=[]), + patch.object(ramalama.common, "podman_machine_accel", False), + patch.object(ramalama.common, "nvidia_selected_devices", list(selected)), + ): + engine.add_device_options() + return engine.exec_args + + def test_cuda_device_options_podman(self): + exec_args = self._cuda_device_args("podman", "0,1") + self.assertEqual(exec_args, ["--device", "nvidia.com/gpu=all", "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + + def test_cuda_device_options_docker(self): + exec_args = self._cuda_device_args("docker", "0,1") + self.assertEqual(exec_args, ["--gpus", "all", "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + + def test_cuda_device_options_podman_selection(self): + # Only the selected GPU is passed in, so it is device 0 in the container + # and every backend sees just that one, Vulkan included. + exec_args = self._cuda_device_args("podman", "1", selected=["1"]) + self.assertEqual(exec_args, ["--device", "nvidia.com/gpu=1", "-e", "CUDA_VISIBLE_DEVICES=0"]) + + def test_cuda_device_options_docker_selection(self): + exec_args = self._cuda_device_args("docker", "1,2", selected=["1", "2"]) + self.assertEqual(exec_args, ["--gpus", '"device=1,2"', "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + def test_add_container_labels(self): args = Namespace(**vars(self.base_args), MODEL="test-model", port="8080", subcommand="run") engine = ramalama.engine.Engine(args) diff --git a/test/unit/test_inference_engine_plugins.py b/test/unit/test_inference_engine_plugins.py index 469e63cc3..0a5198a54 100644 --- a/test/unit/test_inference_engine_plugins.py +++ b/test/unit/test_inference_engine_plugins.py @@ -701,6 +701,40 @@ def test_get_container_image_cuda(self): image = self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") assert image == version_tagged_image("quay.io/ramalama/cuda") + @pytest.mark.parametrize("has_icd,warned", [(True, False), (False, True)]) + def test_get_container_image_vulkan_on_nvidia_warns_without_icd(self, has_icd, warned, monkeypatch): + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.has_nvidia_vulkan_icd", lambda: has_icd) + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "vulkan"}} + config.images.get.return_value = None + config.default_image = version_tagged_image("quay.io/ramalama/ramalama") + with patch("ramalama.plugins.runtimes.inference.llama_cpp.logger.warning") as mock_warning: + self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") + assert mock_warning.called == warned + if warned: + assert "Vulkan ICD" in mock_warning.call_args.args[0] + + def test_get_container_image_cuda_backend_does_not_warn(self, monkeypatch): + # Asking for cuda gets the cuda image, where the ICD is irrelevant. + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.has_nvidia_vulkan_icd", lambda: False) + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "cuda"}} + config.images.get.return_value = None + with patch("ramalama.plugins.runtimes.inference.llama_cpp.logger.warning") as mock_warning: + self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") + mock_warning.assert_not_called() + + def test_get_container_image_vulkan_on_amd_does_not_warn(self, monkeypatch): + # The ICD probe is NVIDIA-specific, so it must not fire for other vendors. + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.has_nvidia_vulkan_icd", lambda: False) + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "auto"}} + config.images.get.return_value = None + config.default_image = version_tagged_image("quay.io/ramalama/ramalama") + with patch("ramalama.plugins.runtimes.inference.llama_cpp.logger.warning") as mock_warning: + self.plugin.get_container_image(config, "HIP_VISIBLE_DEVICES") + mock_warning.assert_not_called() + def test_get_container_image_no_gpu(self): config = MagicMock() config.runtimes = {"llama_cpp": {"backend": "auto"}} @@ -1229,7 +1263,7 @@ def test_mlx_serve_no_generate(self, monkeypatch): # Force backend even with different GPU (warns but allows) ("rocm", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), ("cuda", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), - ("vulkan", "CUDA_VISIBLE_DEVICES", DEFAULT_IMAGE), # Vulkan on NVIDIA (not in preferences, warns) + ("vulkan", "CUDA_VISIBLE_DEVICES", DEFAULT_IMAGE), # Explicit Vulkan on NVIDIA ], ) def test_backend_selection(backend: str, gpu_env: str, expected_result: str, monkeypatch): @@ -1266,9 +1300,11 @@ def test_backend_selection(backend: str, gpu_env: str, expected_result: str, mon ("auto", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), # AMD -> ROCm on Windows ("auto", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), # NVIDIA -> CUDA ("auto", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/intel-gpu")), # Intel -> sycl - # Explicit backends still work + # Explicit backends still work, vulkan included ("vulkan", "HIP_VISIBLE_DEVICES", DEFAULT_IMAGE), ("rocm", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), + ("vulkan", "CUDA_VISIBLE_DEVICES", DEFAULT_IMAGE), + ("cuda", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), ("vulkan", "INTEL_VISIBLE_DEVICES", DEFAULT_IMAGE), ("sycl", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/intel-gpu")), ("openvino", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/openvino")), @@ -1386,7 +1422,7 @@ def test_backend_incompatibility_warning(monkeypatch): "gpu_env,expected_backends", [ ("HIP_VISIBLE_DEVICES", ["auto", "vulkan", "rocm"]), # AMD - ("CUDA_VISIBLE_DEVICES", ["auto", "cuda"]), # NVIDIA + ("CUDA_VISIBLE_DEVICES", ["auto", "cuda", "vulkan"]), # NVIDIA (CUDA preferred) ("INTEL_VISIBLE_DEVICES", ["auto", "vulkan", "sycl", "openvino"]), # Intel (Vulkan preferred) ("ASAHI_VISIBLE_DEVICES", ["auto", "vulkan"]), # Asahi ("ASCEND_VISIBLE_DEVICES", ["auto", "cann"]), # Ascend @@ -1410,7 +1446,7 @@ def test_get_available_backends(gpu_env: Optional[str], expected_backends: list[ "gpu_env,expected_backends", [ ("HIP_VISIBLE_DEVICES", ["auto", "rocm", "vulkan"]), # AMD: ROCm preferred on Windows - ("CUDA_VISIBLE_DEVICES", ["auto", "cuda"]), # NVIDIA: same on all platforms + ("CUDA_VISIBLE_DEVICES", ["auto", "cuda", "vulkan"]), # NVIDIA: same on all platforms ("INTEL_VISIBLE_DEVICES", ["auto", "sycl", "vulkan", "openvino"]), # Intel: sycl preferred on Windows (None, ["auto", "vulkan"]), # No GPU: same on all platforms ], @@ -1453,7 +1489,7 @@ def test_backend_to_gpu_env(self, backend, expected): def test_gpu_backend_preferences_nvidia(self): prefs = get_gpu_backend_preferences("CUDA_VISIBLE_DEVICES") - assert prefs == ["cuda"] + assert prefs == ["cuda", "vulkan"] def test_gpu_backend_preferences_amd(self): prefs = get_gpu_backend_preferences("HIP_VISIBLE_DEVICES")