From 498bde407738eb2fd8d2102f81bf0e3bcc6b994a Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 9 Sep 2026 17:24:32 +0100 Subject: [PATCH 1/7] container-images: install NVIDIA Vulkan ICD deps in the ramalama image NVIDIA's Vulkan ICD is libGLX_nvidia.so.0, injected into the container by the NVIDIA container toolkit. It needs two libraries the image did not carry: libXext.so.6 (a DT_NEEDED of the ICD, which fails loudly) and libEGL.so.1 (resolved internally, which fails silently by returning NULL from vk_icdGetInstanceProcAddr). Without them the loader skips the ICD and falls back to llvmpipe, so llama.cpp runs Vulkan on the CPU. libglvnd-egl provides libEGL.so.1, but its rich dependency on mesa-libEGL would upgrade mesa off MESA_VULKAN_VERSION, so pin mesa-libEGL to the copr version too. Costs ~51 MiB, and lets the ramalama image serve NVIDIA GPUs instead of the multi-GB cuda image. See https://github.com/NVIDIA/nvidia-container-toolkit/issues/191 Signed-off-by: Oliver Walsh --- container-images/scripts/build_llama.sh | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/container-images/scripts/build_llama.sh b/container-images/scripts/build_llama.sh index 72cacb455..5adaddf2c 100755 --- a/container-images/scripts/build_llama.sh +++ b/container-images/scripts/build_llama.sh @@ -142,6 +142,11 @@ dnf_install_runtime_deps() { if [ "$uname_m" = "x86_64" ] || [ "$uname_m" = "aarch64" ]; then dnf copr enable -y slp/mesa-libkrun-vulkan runtime_pkgs+=(vulkan-loader vulkan-tools "mesa-vulkan-drivers-$MESA_VULKAN_VERSION") + # NVIDIA's Vulkan ICD (libGLX_nvidia.so.0, injected by the container + # toolkit) needs libXext and libEGL.so.1 present to initialize. + # libglvnd-egl pulls in mesa-libEGL, so pin it to the copr version to + # avoid dragging the rest of mesa off MESA_VULKAN_VERSION. + runtime_pkgs+=(libXext libglvnd-egl "mesa-libEGL-$MESA_VULKAN_VERSION") else runtime_pkgs+=(openblas) fi From b59a86d5c7eeb8207947ad6102f1801519bbcaa3 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 16 Sep 2026 16:26:06 +0100 Subject: [PATCH 2/7] container-images: request the graphics capability for NVIDIA GPUs NVIDIA_DRIVER_CAPABILITIES gates which driver libraries get injected into the container. The Vulkan ICD only comes in under the "graphics" capability, and the legacy nvidia-container-runtime hook that backs "docker --gpus all" defaults to "compute,utility" when the variable is unset, so the ICD is silently absent under Docker. Set it in the image rather than on the "podman/docker run" command line: the hook reads the variable from the image configuration, so this covers the generated quadlet, kube, compose and stack artifacts too, which build their environment from get_accel_env_vars() and never see anything the engine appends. CDI, which the podman path uses, ignores the variable, so this is a no-op there. Signed-off-by: Oliver Walsh --- container-images/ramalama/Containerfile | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/container-images/ramalama/Containerfile b/container-images/ramalama/Containerfile index d95c24d02..6fb4f9f70 100644 --- a/container-images/ramalama/Containerfile +++ b/container-images/ramalama/Containerfile @@ -8,6 +8,15 @@ RUN ./build_llama.sh ramalama FROM quay.io/fedora/fedora:44 +# NVIDIA_DRIVER_CAPABILITIES gates which driver libraries the NVIDIA container +# toolkit injects. The Vulkan ICD only comes in under the "graphics" capability, +# and the legacy nvidia-container-runtime hook that backs "docker --gpus all" +# defaults to "compute,utility" when the variable is unset, so the ICD is +# silently absent under Docker. The hook reads the variable from the image, so +# setting it here covers "ramalama run" and the generated quadlet, kube, compose +# and stack artifacts alike. CDI, which the podman path uses, ignores it. +ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility,graphics + RUN --mount=type=bind,from=builder,source=/tmp/install,target=/tmp/install,relabel=private \ cp -a /tmp/install/bin/ /usr/ && \ cp -a /tmp/install/lib64/*.so* /usr/lib64/ From 717599f887467e14f20fc48907b05f1c2988f6d2 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 16 Sep 2026 16:34:36 +0100 Subject: [PATCH 3/7] engine: pass only the selected NVIDIA GPUs into the container A narrowed CUDA_VISIBLE_DEVICES was honoured by making every GPU visible to the container and letting the CUDA runtime inside filter on the variable. Only the cuda backend does that filtering: llama.cpp's Vulkan backend indexes Vulkan's own device enumeration and never reads it, so the selection is silently ignored and inference lands on whichever GPU Vulkan enumerates first. Select at the device level instead, so the container cannot see the GPUs that were excluded, and renumber the variable to match what is left. The device names come from find_in_cdi(), so they are known to be in the CDI configuration; where the narrowing cannot be expressed that way (only an "all" device is configured) behaviour is unchanged. Signed-off-by: Oliver Walsh --- ramalama/common.py | 36 ++++++++++++++++++++++--- ramalama/engine.py | 26 ++++++++++++++++-- test/unit/test_common.py | 58 ++++++++++++++++++++++++++++++++++++++++ test/unit/test_engine.py | 31 +++++++++++++++++++++ 4 files changed, 146 insertions(+), 5 deletions(-) diff --git a/ramalama/common.py b/ramalama/common.py index 2987d6cb1..1ffdf1329 100644 --- a/ramalama/common.py +++ b/ramalama/common.py @@ -57,6 +57,22 @@ def sanitize_filename(filename: str) -> str: podman_machine_accel = False +# CDI device names of the NVIDIA GPUs the user narrowed CUDA_VISIBLE_DEVICES +# down to, set by check_nvidia(). Empty when every detected GPU is in play. +nvidia_selected_devices: list[str] = [] + + +def container_cuda_visible_devices(value: str) -> str: + """CUDA_VISIBLE_DEVICES as it should read inside the container. + + Where the selection is narrowed, only those GPUs are passed in and the + container numbers them from zero, so the host's indices no longer describe + them. Returns the value unchanged when every GPU is in play. + """ + if not nvidia_selected_devices: + return value + return ",".join(str(i) for i in range(len(nvidia_selected_devices))) + def confirm_no_gpu(name, provider) -> bool: while True: @@ -438,9 +454,13 @@ def find_in_cdi(devices: list[str]) -> tuple[list[str], list[str]]: for device in devices: if device in cdi_device_names: configured.append(device) - # A device can be specified by a prefix of the uuid - elif device.startswith("GPU") and any(name.startswith(device) for name in cdi_device_names): - configured.append(device) + # A device can be specified by a prefix of the uuid. Record the full + # name it resolves to: it reaches "--device nvidia.com/gpu=", + # which matches the CDI configuration exactly and not by prefix. + elif device.startswith("GPU") and ( + full_name := next((name for name in cdi_device_names if name.startswith(device)), None) + ): + configured.append(full_name) else: perror(f"Device {device} does not have a CDI configuration") unconfigured.append(device) @@ -515,6 +535,16 @@ def check_nvidia() -> Optional[Literal["cuda"]]: if not configured: configured = indices + # Record a narrowed selection so the engine can pass just those GPUs + # into the container. Not every backend filters on CUDA_VISIBLE_DEVICES + # (llama.cpp's Vulkan backend indexes Vulkan's own device enumeration), + # so the selection has to happen at the device level to be honoured. + # These names came back from find_in_cdi(), so they are known to the CDI + # configuration; the fallback above to every index has not been checked + # and stays on the "all" device. + global nvidia_selected_devices + nvidia_selected_devices = configured if set(configured) != set(indices) else [] + os.environ["CUDA_VISIBLE_DEVICES"] = ','.join(configured) return "cuda" diff --git a/ramalama/engine.py b/ramalama/engine.py index 90aa125eb..01ffdadc1 100644 --- a/ramalama/engine.py +++ b/ramalama/engine.py @@ -15,7 +15,17 @@ # Live reference for checking global vars import ramalama.common from ramalama.arg_types import BaseEngineArgsType -from ramalama.common import check_nvidia, engine_cmd, exec_cmd, genname, get_accel_env_vars, host_path, perror, run_cmd +from ramalama.common import ( + check_nvidia, + container_cuda_visible_devices, + engine_cmd, + exec_cmd, + genname, + get_accel_env_vars, + host_path, + perror, + run_cmd, +) from ramalama.compat import NamedTemporaryFile from ramalama.config import ActiveConfig from ramalama.host_utils import ( @@ -118,11 +128,23 @@ def add_device_options(self): for k, v in get_accel_env_vars().items(): # Special case for Cuda if k == "CUDA_VISIBLE_DEVICES": + # Pass in only the GPUs the user selected rather than all of + # them plus CUDA_VISIBLE_DEVICES to filter with: the Vulkan + # backend never reads that variable, it indexes Vulkan's own + # device enumeration, so a narrowed selection is only honoured + # if the container cannot see the other GPUs in the first place. + selected = ramalama.common.nvidia_selected_devices if self.use_docker: - self.exec_args += ["--gpus", "all"] + # docker's csv parser needs the quotes to keep a + # comma-separated device list in one field. + self.exec_args += ["--gpus", f'"device={",".join(selected)}"' if selected else "all"] + elif selected: + for name in selected: + self.exec_args += ["--device", f"nvidia.com/gpu={name}"] else: # newer Podman versions support --gpus=all, but < 5.0 do not self.exec_args += ["--device", "nvidia.com/gpu=all"] + v = container_cuda_visible_devices(v) elif k == "MUSA_VISIBLE_DEVICES": self.exec_args += ["--env", "MTHREADS_VISIBLE_DEVICES=all"] elif k == "INTEL_VISIBLE_DEVICES": diff --git a/test/unit/test_common.py b/test/unit/test_common.py index f914059c4..0fa7b0dd5 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -9,6 +9,7 @@ import pytest +import ramalama.common from ramalama.cli import ( default_image, default_rag_image, @@ -20,6 +21,7 @@ accel_image, check_intel, check_nvidia, + container_cuda_visible_devices, engine_cmd, ensure_image, find_in_cdi, @@ -287,6 +289,14 @@ def test_pull_fails_ramalama_image_fallback_fails_raises(self, mock_run_cmd): class TestCheckNvidia: def setup_method(self): check_nvidia.cache_clear() + self.cuda_visible_devices = os.environ.pop("CUDA_VISIBLE_DEVICES", None) + ramalama.common.nvidia_selected_devices = [] + + def teardown_method(self): + os.environ.pop("CUDA_VISIBLE_DEVICES", None) + if self.cuda_visible_devices is not None: + os.environ["CUDA_VISIBLE_DEVICES"] = self.cuda_visible_devices + ramalama.common.nvidia_selected_devices = [] @patch("ramalama.common.find_in_cdi") @patch("ramalama.common.run_cmd") @@ -318,6 +328,51 @@ def test_check_nvidia_no_cdi_toolkit_present(self, mock_run_cmd, mock_find_in_cd printed = " ".join(str(c.args[0]) for c in mock_perror.call_args_list) assert "nvidia-ctk cdi generate" in printed + @patch("ramalama.common.find_in_cdi") + @patch("ramalama.common.run_cmd") + def test_check_nvidia_all_gpus_are_not_a_selection(self, mock_run_cmd, mock_find_in_cdi): + mock_find_in_cdi.return_value = (["all"], []) + mock_run_cmd.return_value.stdout = "0,GPU-1111\n1,GPU-2222" + assert check_nvidia() == "cuda" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "0,1" + assert ramalama.common.nvidia_selected_devices == [] + + @patch("ramalama.common.find_in_cdi") + @patch("ramalama.common.run_cmd") + def test_check_nvidia_records_narrowed_selection(self, mock_run_cmd, mock_find_in_cdi): + os.environ["CUDA_VISIBLE_DEVICES"] = "1" + mock_find_in_cdi.return_value = (["1", "all"], []) + mock_run_cmd.return_value.stdout = "0,GPU-1111\n1,GPU-2222" + assert check_nvidia() == "cuda" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1" + assert ramalama.common.nvidia_selected_devices == ["1"] + + @patch("ramalama.common.find_in_cdi") + @patch("ramalama.common.run_cmd") + def test_check_nvidia_selection_needs_a_cdi_device(self, mock_run_cmd, mock_find_in_cdi): + # Only the "all" device is configured, so the narrowing cannot be + # expressed as a device and every GPU stays visible, as before. + os.environ["CUDA_VISIBLE_DEVICES"] = "1" + mock_find_in_cdi.return_value = (["all"], ["1"]) + mock_run_cmd.return_value.stdout = "0,GPU-1111\n1,GPU-2222" + assert check_nvidia() == "cuda" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "0,1" + assert ramalama.common.nvidia_selected_devices == [] + + @pytest.mark.parametrize( + "selected,value,expected", + [ + # Every GPU is in play, so the host's numbering still describes them. + ([], "0,1", "0,1"), + (["1"], "1", "0"), + (["1", "2"], "1,2", "0,1"), + (["GPU-2222"], "GPU-2222", "0"), + ], + ) + def test_container_cuda_visible_devices(self, selected, value, expected): + with patch.object(ramalama.common, "nvidia_selected_devices", selected): + assert container_cuda_visible_devices(value) == expected + @patch("ramalama.common.run_cmd") def test_check_nvidia_smi_failure(self, mock_run_cmd): mock_run_cmd.side_effect = subprocess.CalledProcessError(1, "nvidia-smi") @@ -610,6 +665,9 @@ def open_side_effect(path, *args, **kwargs): (["all"], ["all"], []), (["0", "all"], ["0", "all"], []), ([CDI_GPU_UUID, "all"], [CDI_GPU_UUID, "all"], []), + # An abbreviated uuid resolves to the full CDI device name, which is + # what "--device nvidia.com/gpu=" needs. + ([CDI_GPU_UUID[:12], "all"], [CDI_GPU_UUID, "all"], []), (["1", "all"], ["all"], ["1"]), (["dummy", "all"], ["all"], ["dummy"]), ], diff --git a/test/unit/test_engine.py b/test/unit/test_engine.py index b86d16bc0..7f2b0c78e 100644 --- a/test/unit/test_engine.py +++ b/test/unit/test_engine.py @@ -27,6 +27,37 @@ def test_init_basic(self): self.assertEqual(engine.use_podman, True) self.assertEqual(engine.use_docker, False) + def _cuda_device_args(self, engine_name, visible_devices, selected=()): + args = Namespace(**{**vars(self.base_args), "engine": engine_name}) + engine = ramalama.engine.Engine(args) + engine.exec_args = [] + with ( + patch("ramalama.engine.get_accel_env_vars", return_value={"CUDA_VISIBLE_DEVICES": visible_devices}), + patch("glob.glob", return_value=[]), + patch.object(ramalama.common, "podman_machine_accel", False), + patch.object(ramalama.common, "nvidia_selected_devices", list(selected)), + ): + engine.add_device_options() + return engine.exec_args + + def test_cuda_device_options_podman(self): + exec_args = self._cuda_device_args("podman", "0,1") + self.assertEqual(exec_args, ["--device", "nvidia.com/gpu=all", "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + + def test_cuda_device_options_docker(self): + exec_args = self._cuda_device_args("docker", "0,1") + self.assertEqual(exec_args, ["--gpus", "all", "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + + def test_cuda_device_options_podman_selection(self): + # Only the selected GPU is passed in, so it is device 0 in the container + # and every backend sees just that one, Vulkan included. + exec_args = self._cuda_device_args("podman", "1", selected=["1"]) + self.assertEqual(exec_args, ["--device", "nvidia.com/gpu=1", "-e", "CUDA_VISIBLE_DEVICES=0"]) + + def test_cuda_device_options_docker_selection(self): + exec_args = self._cuda_device_args("docker", "1,2", selected=["1", "2"]) + self.assertEqual(exec_args, ["--gpus", '"device=1,2"', "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + def test_add_container_labels(self): args = Namespace(**vars(self.base_args), MODEL="test-model", port="8080", subcommand="run") engine = ramalama.engine.Engine(args) From 713cd8130fa7a0bbe2a304a2d8a7a90d45e7b239 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Thu, 17 Sep 2026 13:09:12 +0100 Subject: [PATCH 4/7] cli: resolve the container image after --backend is parsed The --image default is accel_image(), evaluated while the parser is built and so before --backend has been seen. It is the image for the backend that auto resolves to, which is not necessarily the one the command will run. That default is then copied into the config like any other argument, but only where it differs from the value already there, and whether it differs decides which of two ways the image comes out wrong. For a detected NVIDIA GPU it does, so the image counts as explicitly set from then on and every later accel_image() call short-circuits on it, leaving the backend no say. Where it does not, args.image keeps the stale value instead, which is the image --dryrun prints and --generate writes out. The same comparison drops an --image the user passed whenever it matches the default, so the default image cannot be asked for by name on a host where a GPU would otherwise select another one. Go by whether the option was passed rather than by the value it carries, and resolve the image once more after the runtime config has picked up --backend. Signed-off-by: Oliver Walsh --- ramalama/cli.py | 27 +++++++++++++++++++++++++++ test/unit/test_common.py | 32 ++++++++++++++++++++++++++++++++ 2 files changed, 59 insertions(+) diff --git a/ramalama/cli.py b/ramalama/cli.py index 1758f30d3..2a149c6f3 100644 --- a/ramalama/cli.py +++ b/ramalama/cli.py @@ -125,6 +125,15 @@ def parse_port_option(option: str) -> str: class OverrideDefaultAction(argparse.Action): + # Options whose default is computed while the parser is built, before the + # rest of the command line has been seen. Recorded so such a default can be + # told apart from a value the user passed. + dests: set[str] = set() + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + OverrideDefaultAction.dests.add(self.dest) + def __call__(self, parser, namespace, values, option_string=None): setattr(namespace, self.dest, values) setattr(namespace, self.dest + '_override', True) @@ -221,11 +230,29 @@ def parse_args_from_cmd(cmd: list[str]) -> tuple[argparse.ArgumentParser, argpar post_parse_setup(args) for arg in args.__dict__.keys() & config._fields: + if arg in OverrideDefaultAction.dests: + # These options record whether they were passed, so go by that rather + # than by the value. A computed default is not the user's choice and + # does not belong in the config: --image defaults to the image the + # detected GPU resolves to, worked out before --backend was parsed, + # and storing it marks the image as chosen rather than derived. A + # value the user did pass belongs there even where it matches the + # default, which is how "--image $RAMALAMA_DEFAULT_IMAGE" asks for + # that image instead of the one the GPU would select. + if getattr(args, f"{arg}_override", False): + setattr(config, arg, getattr(args, arg)) + continue if getattr(args, arg) != getattr(config, arg): setattr(config, arg, getattr(args, arg)) runtime_plugin = get_runtime(config.runtime) runtime_plugin.sync_args_to_runtime_config(args, config) + + # The runtime config now carries --backend, so the image the command will + # run can be resolved for real. + if hasattr(args, "image") and not getattr(args, "image_override", False): + args.image = accel_image(config) + return parser, args diff --git a/test/unit/test_common.py b/test/unit/test_common.py index 0fa7b0dd5..2812f0841 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -34,6 +34,7 @@ populate_volume_from_image, rm_until_substring, verify_checksum, + version_tagged_image, ) from ramalama.compat import NamedTemporaryFile from ramalama.config import DEFAULT_IMAGE, load_config @@ -139,6 +140,9 @@ def test_verify_checksum( ("HIP_VISIBLE_DEVICES", f"{_BASE_IMAGE}:latest", None, None, f"{_BASE_IMAGE}:latest"), ("HIP_VISIBLE_DEVICES", None, f"{_BASE_IMAGE}:latest", None, f"{_BASE_IMAGE}:latest"), ("HIP_VISIBLE_DEVICES", None, None, f"{_BASE_IMAGE}:latest", f"{_BASE_IMAGE}:latest"), + # An --image that happens to match the default is still the user asking + # for it, and wins over the image the detected GPU would select. + ("CUDA_VISIBLE_DEVICES", None, None, DEFAULT_IMAGE, DEFAULT_IMAGE), ], ) def test_accel_image( @@ -178,6 +182,34 @@ def test_accel_image( assert accel_image(config) == expected_result +@pytest.mark.parametrize( + "accel_env,backend,expected_result", + [ + # Left on auto, the detected GPU picks the image. + ("CUDA_VISIBLE_DEVICES", "auto", version_tagged_image("quay.io/ramalama/cuda")), + ("CUDA_VISIBLE_DEVICES", "cuda", version_tagged_image("quay.io/ramalama/cuda")), + # A backend the user asked for wins over the detected GPU, even though + # --image defaults to the image that GPU resolves to. + ("HIP_VISIBLE_DEVICES", "auto", DEFAULT_IMAGE), + ("HIP_VISIBLE_DEVICES", "vulkan", DEFAULT_IMAGE), + ("HIP_VISIBLE_DEVICES", "rocm", version_tagged_image("quay.io/ramalama/rocm")), + ], +) +def test_accel_image_follows_backend(accel_env: str, backend: str, expected_result: str, monkeypatch): + monkeypatch.setattr("ramalama.common.get_accel", lambda: "none") + + env = {"RAMALAMA_CONFIG": "/dev/null", accel_env: "1"} + with patch.dict("os.environ", env, clear=True): + config = load_config() + with patch("ramalama.cli.ActiveConfig", return_value=config): + default_image.cache_clear() + default_rag_image.cache_clear() + default_tools_image.cache_clear() + _, args = parse_args_from_cmd(["run", "--backend", backend, "granite"]) + assert accel_image(config) == expected_result + assert args.image == expected_result + + @patch("ramalama.common.run_cmd") @patch("ramalama.common.handle_provider") def test_apple_vm_returns_result_podman_v5(mock_handle_provider, mock_run_cmd): From 957f8e75c549c94c1e2e2ce1a2a6e4ec570335a7 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 16 Sep 2026 17:21:27 +0100 Subject: [PATCH 5/7] llama.cpp: make the vulkan backend selectable on NVIDIA GPUs llama.cpp's Vulkan backend runs on NVIDIA hardware through the vendor's own ICD, which the container toolkit injects, so the small ramalama image can serve an NVIDIA GPU. Verified enumerating both GPUs on a 2x RTX 3090 host. Until now the backend list for NVIDIA held cuda alone, which also made it the only value --backend would accept there, since the choices come from the same list. Add vulkan to it, behind cuda: auto still resolves to cuda and nothing changes by default, but "--backend vulkan" is now a supported configuration on NVIDIA rather than a rejected one. Signed-off-by: Oliver Walsh --- docs/options/backend.md | 4 ++-- docs/ramalama-bench.1.md | 4 ++-- docs/ramalama-perplexity.1.md | 4 ++-- docs/ramalama-run.1.md | 4 ++-- docs/ramalama-sandbox-goose.1.md | 4 ++-- docs/ramalama-sandbox-opencode.1.md | 4 ++-- docs/ramalama-sandbox-pi.1.md | 4 ++-- docs/ramalama-serve.1.md | 4 ++-- docs/ramalama.conf | 4 ++-- docs/ramalama.conf.5.md | 4 ++-- ramalama/plugins/runtimes/inference/llama_cpp.py | 2 +- test/unit/test_common.py | 1 + test/unit/test_inference_engine_plugins.py | 12 +++++++----- 13 files changed, 29 insertions(+), 26 deletions(-) diff --git a/docs/options/backend.md b/docs/options/backend.md index 2b9c9d70b..41d6a40da 100644 --- a/docs/options/backend.md +++ b/docs/options/backend.md @@ -10,7 +10,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -21,7 +21,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-bench.1.md b/docs/ramalama-bench.1.md index 6b77633e7..f82372170 100644 --- a/docs/ramalama-bench.1.md +++ b/docs/ramalama-bench.1.md @@ -46,7 +46,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -57,7 +57,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-perplexity.1.md b/docs/ramalama-perplexity.1.md index bb6833404..6cff430ba 100644 --- a/docs/ramalama-perplexity.1.md +++ b/docs/ramalama-perplexity.1.md @@ -46,7 +46,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -57,7 +57,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-run.1.md b/docs/ramalama-run.1.md index 059c97e4b..5ed45f709 100644 --- a/docs/ramalama-run.1.md +++ b/docs/ramalama-run.1.md @@ -58,7 +58,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -69,7 +69,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-sandbox-goose.1.md b/docs/ramalama-sandbox-goose.1.md index 85c232d64..35c1fd3af 100644 --- a/docs/ramalama-sandbox-goose.1.md +++ b/docs/ramalama-sandbox-goose.1.md @@ -54,7 +54,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -65,7 +65,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-sandbox-opencode.1.md b/docs/ramalama-sandbox-opencode.1.md index 379b1c843..5af6bf194 100644 --- a/docs/ramalama-sandbox-opencode.1.md +++ b/docs/ramalama-sandbox-opencode.1.md @@ -54,7 +54,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -65,7 +65,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-sandbox-pi.1.md b/docs/ramalama-sandbox-pi.1.md index 396508d20..868af0d3e 100644 --- a/docs/ramalama-sandbox-pi.1.md +++ b/docs/ramalama-sandbox-pi.1.md @@ -58,7 +58,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -69,7 +69,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama-serve.1.md b/docs/ramalama-serve.1.md index 1c604a1c0..83400ba02 100644 --- a/docs/ramalama-serve.1.md +++ b/docs/ramalama-serve.1.md @@ -87,7 +87,7 @@ Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: - **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) -- **NVIDIA GPUs**: cuda +- **NVIDIA GPUs**: cuda; vulkan available as explicit option - **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa @@ -98,7 +98,7 @@ Available backends depend on the detected GPU hardware. - On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred **Explicit backend selection**: -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama.conf b/docs/ramalama.conf index 4390b3b45..cc54ea239 100644 --- a/docs/ramalama.conf +++ b/docs/ramalama.conf @@ -241,12 +241,12 @@ # Valid options: auto, vulkan, rocm, cuda, sycl, openvino, cann, musa # - auto (default): Automatically selects the preferred backend based on detected GPU # - AMD GPUs: vulkan (Linux/macOS) or rocm (Windows) -# - NVIDIA GPUs: cuda +# - NVIDIA GPUs: cuda; vulkan available as explicit option # - Intel GPUs: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option # - Ascend NPUs: cann # - MUSA GPUs: musa # - No GPU: vulkan (CPU fallback) -# - vulkan: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +# - vulkan: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) # - rocm: Use AMD ROCm backend (AMD GPUs only) # - cuda: Use NVIDIA CUDA backend (NVIDIA GPUs only) # - sycl: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/docs/ramalama.conf.5.md b/docs/ramalama.conf.5.md index 641304153..529913a09 100644 --- a/docs/ramalama.conf.5.md +++ b/docs/ramalama.conf.5.md @@ -240,13 +240,13 @@ Valid options: `auto`, `vulkan`, `rocm`, `cuda`, `sycl`, `openvino`, `cann`, `mu - **auto** (default): Automatically selects the preferred backend based on detected GPU: - AMD GPUs: vulkan (Linux/macOS) or rocm (Windows) - - NVIDIA GPUs: cuda + - NVIDIA GPUs: cuda; vulkan available as explicit option - Intel GPUs: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option - Ascend NPUs: cann - MUSA GPUs: musa - No GPU: vulkan (CPU fallback) -- **vulkan**: Use Vulkan-based inference (compatible with AMD, Intel, and CPU) +- **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) - **rocm**: Use AMD ROCm backend (AMD GPUs only) - **cuda**: Use NVIDIA CUDA backend (NVIDIA GPUs only) - **sycl**: Use Intel SYCL/oneAPI backend (Intel GPUs only) diff --git a/ramalama/plugins/runtimes/inference/llama_cpp.py b/ramalama/plugins/runtimes/inference/llama_cpp.py index 97a8fefe5..0004fef29 100644 --- a/ramalama/plugins/runtimes/inference/llama_cpp.py +++ b/ramalama/plugins/runtimes/inference/llama_cpp.py @@ -144,7 +144,7 @@ def get_gpu_backend_preferences(gpu_type: str) -> list[str]: preferences = { "HIP_VISIBLE_DEVICES": ["vulkan", "rocm"], # AMD: Vulkan preferred - "CUDA_VISIBLE_DEVICES": ["cuda"], # NVIDIA: CUDA only + "CUDA_VISIBLE_DEVICES": ["cuda", "vulkan"], # NVIDIA: CUDA preferred "INTEL_VISIBLE_DEVICES": ["vulkan", "sycl", "openvino"], # Intel: Vulkan preferred "ASAHI_VISIBLE_DEVICES": ["vulkan"], # Asahi: Vulkan only "ASCEND_VISIBLE_DEVICES": ["cann"], # Ascend: CANN only diff --git a/test/unit/test_common.py b/test/unit/test_common.py index 2812f0841..d6f6a65f1 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -190,6 +190,7 @@ def test_accel_image( ("CUDA_VISIBLE_DEVICES", "cuda", version_tagged_image("quay.io/ramalama/cuda")), # A backend the user asked for wins over the detected GPU, even though # --image defaults to the image that GPU resolves to. + ("CUDA_VISIBLE_DEVICES", "vulkan", DEFAULT_IMAGE), ("HIP_VISIBLE_DEVICES", "auto", DEFAULT_IMAGE), ("HIP_VISIBLE_DEVICES", "vulkan", DEFAULT_IMAGE), ("HIP_VISIBLE_DEVICES", "rocm", version_tagged_image("quay.io/ramalama/rocm")), diff --git a/test/unit/test_inference_engine_plugins.py b/test/unit/test_inference_engine_plugins.py index 469e63cc3..fe6836736 100644 --- a/test/unit/test_inference_engine_plugins.py +++ b/test/unit/test_inference_engine_plugins.py @@ -1229,7 +1229,7 @@ def test_mlx_serve_no_generate(self, monkeypatch): # Force backend even with different GPU (warns but allows) ("rocm", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), ("cuda", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), - ("vulkan", "CUDA_VISIBLE_DEVICES", DEFAULT_IMAGE), # Vulkan on NVIDIA (not in preferences, warns) + ("vulkan", "CUDA_VISIBLE_DEVICES", DEFAULT_IMAGE), # Explicit Vulkan on NVIDIA ], ) def test_backend_selection(backend: str, gpu_env: str, expected_result: str, monkeypatch): @@ -1266,9 +1266,11 @@ def test_backend_selection(backend: str, gpu_env: str, expected_result: str, mon ("auto", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), # AMD -> ROCm on Windows ("auto", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), # NVIDIA -> CUDA ("auto", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/intel-gpu")), # Intel -> sycl - # Explicit backends still work + # Explicit backends still work, vulkan included ("vulkan", "HIP_VISIBLE_DEVICES", DEFAULT_IMAGE), ("rocm", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), + ("vulkan", "CUDA_VISIBLE_DEVICES", DEFAULT_IMAGE), + ("cuda", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), ("vulkan", "INTEL_VISIBLE_DEVICES", DEFAULT_IMAGE), ("sycl", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/intel-gpu")), ("openvino", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/openvino")), @@ -1386,7 +1388,7 @@ def test_backend_incompatibility_warning(monkeypatch): "gpu_env,expected_backends", [ ("HIP_VISIBLE_DEVICES", ["auto", "vulkan", "rocm"]), # AMD - ("CUDA_VISIBLE_DEVICES", ["auto", "cuda"]), # NVIDIA + ("CUDA_VISIBLE_DEVICES", ["auto", "cuda", "vulkan"]), # NVIDIA (CUDA preferred) ("INTEL_VISIBLE_DEVICES", ["auto", "vulkan", "sycl", "openvino"]), # Intel (Vulkan preferred) ("ASAHI_VISIBLE_DEVICES", ["auto", "vulkan"]), # Asahi ("ASCEND_VISIBLE_DEVICES", ["auto", "cann"]), # Ascend @@ -1410,7 +1412,7 @@ def test_get_available_backends(gpu_env: Optional[str], expected_backends: list[ "gpu_env,expected_backends", [ ("HIP_VISIBLE_DEVICES", ["auto", "rocm", "vulkan"]), # AMD: ROCm preferred on Windows - ("CUDA_VISIBLE_DEVICES", ["auto", "cuda"]), # NVIDIA: same on all platforms + ("CUDA_VISIBLE_DEVICES", ["auto", "cuda", "vulkan"]), # NVIDIA: same on all platforms ("INTEL_VISIBLE_DEVICES", ["auto", "sycl", "vulkan", "openvino"]), # Intel: sycl preferred on Windows (None, ["auto", "vulkan"]), # No GPU: same on all platforms ], @@ -1453,7 +1455,7 @@ def test_backend_to_gpu_env(self, backend, expected): def test_gpu_backend_preferences_nvidia(self): prefs = get_gpu_backend_preferences("CUDA_VISIBLE_DEVICES") - assert prefs == ["cuda"] + assert prefs == ["cuda", "vulkan"] def test_gpu_backend_preferences_amd(self): prefs = get_gpu_backend_preferences("HIP_VISIBLE_DEVICES") From 8fcb601945efec1f469fc6a935dcfee646829571 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 16 Sep 2026 17:11:01 +0100 Subject: [PATCH 6/7] llama.cpp: warn when the host has no NVIDIA Vulkan ICD Vulkan is only usable on an NVIDIA GPU because the container toolkit injects the vendor ICD. Where it does not - an old toolkit, a CDI spec generated without the graphics libraries, nouveau instead of the proprietary driver - llama.cpp finds only mesa's llvmpipe and serves from the CPU. That is orders of magnitude slower and never errors, unlike the cuda image, which fails loudly. Probe the host for an *nvidia*.json ICD manifest and warn once when the vulkan backend is resolved for an NVIDIA GPU without one, pointing at the container toolkit and at --backend cuda. Signed-off-by: Oliver Walsh --- ramalama/common.py | 14 ++++++++ .../plugins/runtimes/inference/llama_cpp.py | 21 ++++++++++++ test/unit/conftest.py | 14 ++++++++ test/unit/test_common.py | 15 ++++++++ test/unit/test_inference_engine_plugins.py | 34 +++++++++++++++++++ 5 files changed, 98 insertions(+) diff --git a/ramalama/common.py b/ramalama/common.py index 1ffdf1329..fa38f08bf 100644 --- a/ramalama/common.py +++ b/ramalama/common.py @@ -488,6 +488,20 @@ def check_metal(args: ContainerArgType) -> bool: return platform.system() == "Darwin" +@lru_cache(maxsize=1) +def has_nvidia_vulkan_icd() -> bool: + """True when NVIDIA's Vulkan ICD manifest is installed on the host. + + The container toolkit injects the ICD from the host driver installation, so + when it is missing here the vulkan backend finds only mesa's llvmpipe inside + the container and runs on the CPU instead of failing. + """ + return any( + glob.glob(os.path.join(host_path(icd_dir), "*nvidia*.json")) + for icd_dir in ("/usr/share/vulkan/icd.d", "/etc/vulkan/icd.d") + ) + + @lru_cache(maxsize=1) def check_nvidia() -> Optional[Literal["cuda"]]: try: diff --git a/ramalama/plugins/runtimes/inference/llama_cpp.py b/ramalama/plugins/runtimes/inference/llama_cpp.py index 0004fef29..0a4a537e6 100644 --- a/ramalama/plugins/runtimes/inference/llama_cpp.py +++ b/ramalama/plugins/runtimes/inference/llama_cpp.py @@ -12,6 +12,7 @@ from collections.abc import Callable, Mapping from dataclasses import asdict, dataclass, field from datetime import datetime, timezone +from functools import lru_cache from http.client import HTTPConnection from typing import Any, Literal, Optional, get_args from urllib.parse import urlparse @@ -40,6 +41,7 @@ ensure_image, genname, get_gpu_type_env_vars, + has_nvidia_vulkan_icd, run_cmd, set_accel_env_vars, set_gpu_type_env_vars, @@ -159,6 +161,22 @@ def get_gpu_backend_preferences(gpu_type: str) -> list[str]: return preferences.get(gpu_type, []) +@lru_cache(maxsize=1) +def warn_without_nvidia_vulkan_icd() -> None: + """Warn once when vulkan is picked for an NVIDIA GPU the host cannot drive. + + Vulkan is only usable on NVIDIA because the container toolkit injects the + vendor ICD. Without it llama.cpp silently serves from the CPU, which is far + slower but never fails, so say so up front.""" + if has_nvidia_vulkan_icd(): + return + logger.warning( + "No NVIDIA Vulkan ICD found in /usr/share/vulkan/icd.d or /etc/vulkan/icd.d. " + "Inference may fall back to the CPU. Install the nvidia-container-toolkit, which " + "provides the ICD for the container, or select the cuda backend with --backend cuda." + ) + + def backend_to_gpu_env(backend: str) -> str: """Maps a backend name to its corresponding GPU environment variable.""" mapping = { @@ -353,6 +371,9 @@ def get_container_image(self, config: Any, detected_gpu_type: str) -> Optional[s f"Recommended backends for {gpu_name}: {', '.join(preferences)}" ) + if gpu_type == "GGML_VK_VISIBLE_DEVICES" and detected_gpu_type == "CUDA_VISIBLE_DEVICES": + warn_without_nvidia_vulkan_icd() + override = config.images.get(gpu_type) if gpu_type else None return override if override else _LLAMA_CPP_IMAGES.get(gpu_type, config.default_image) diff --git a/test/unit/conftest.py b/test/unit/conftest.py index 3adaacb17..50d6bb225 100644 --- a/test/unit/conftest.py +++ b/test/unit/conftest.py @@ -45,6 +45,20 @@ def _isolate_from_toolbox(): in_toolbox.cache_clear() +@pytest.fixture(autouse=True) +def _clear_vulkan_icd_cache(): + """The ICD probe and the warning it drives are cached for the process, so + clear them around every test rather than carrying the host's answer.""" + from ramalama.common import has_nvidia_vulkan_icd + from ramalama.plugins.runtimes.inference.llama_cpp import warn_without_nvidia_vulkan_icd + + for cached in (has_nvidia_vulkan_icd, warn_without_nvidia_vulkan_icd): + cached.cache_clear() + yield + for cached in (has_nvidia_vulkan_icd, warn_without_nvidia_vulkan_icd): + cached.cache_clear() + + @pytest.fixture def force_oci_image(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr(OCIStrategyFactory, "resolve", lambda self, model: self.strategies("image")) diff --git a/test/unit/test_common.py b/test/unit/test_common.py index d6f6a65f1..cd074e1b5 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -26,6 +26,7 @@ ensure_image, find_in_cdi, get_accel, + has_nvidia_vulkan_icd, host_available, host_cmd, host_path, @@ -949,3 +950,17 @@ def test_host_path_in_toolbox_no_run_host(self): patch("os.path.isdir", return_value=False), ): assert host_path("/etc/cdi") == "/etc/cdi" + + +class TestHasNvidiaVulkanIcd: + @pytest.mark.parametrize("icd_dir", ["/usr/share/vulkan/icd.d", "/etc/vulkan/icd.d"]) + def test_icd_installed(self, icd_dir): + with patch( + "ramalama.common.glob.glob", + side_effect=lambda p: [f"{icd_dir}/nvidia_icd.json"] if p.startswith(icd_dir) else [], + ): + assert has_nvidia_vulkan_icd() + + def test_only_other_vendors(self): + with patch("ramalama.common.glob.glob", return_value=[]): + assert not has_nvidia_vulkan_icd() diff --git a/test/unit/test_inference_engine_plugins.py b/test/unit/test_inference_engine_plugins.py index fe6836736..0a5198a54 100644 --- a/test/unit/test_inference_engine_plugins.py +++ b/test/unit/test_inference_engine_plugins.py @@ -701,6 +701,40 @@ def test_get_container_image_cuda(self): image = self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") assert image == version_tagged_image("quay.io/ramalama/cuda") + @pytest.mark.parametrize("has_icd,warned", [(True, False), (False, True)]) + def test_get_container_image_vulkan_on_nvidia_warns_without_icd(self, has_icd, warned, monkeypatch): + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.has_nvidia_vulkan_icd", lambda: has_icd) + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "vulkan"}} + config.images.get.return_value = None + config.default_image = version_tagged_image("quay.io/ramalama/ramalama") + with patch("ramalama.plugins.runtimes.inference.llama_cpp.logger.warning") as mock_warning: + self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") + assert mock_warning.called == warned + if warned: + assert "Vulkan ICD" in mock_warning.call_args.args[0] + + def test_get_container_image_cuda_backend_does_not_warn(self, monkeypatch): + # Asking for cuda gets the cuda image, where the ICD is irrelevant. + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.has_nvidia_vulkan_icd", lambda: False) + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "cuda"}} + config.images.get.return_value = None + with patch("ramalama.plugins.runtimes.inference.llama_cpp.logger.warning") as mock_warning: + self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") + mock_warning.assert_not_called() + + def test_get_container_image_vulkan_on_amd_does_not_warn(self, monkeypatch): + # The ICD probe is NVIDIA-specific, so it must not fire for other vendors. + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.has_nvidia_vulkan_icd", lambda: False) + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "auto"}} + config.images.get.return_value = None + config.default_image = version_tagged_image("quay.io/ramalama/ramalama") + with patch("ramalama.plugins.runtimes.inference.llama_cpp.logger.warning") as mock_warning: + self.plugin.get_container_image(config, "HIP_VISIBLE_DEVICES") + mock_warning.assert_not_called() + def test_get_container_image_no_gpu(self): config = MagicMock() config.runtimes = {"llama_cpp": {"backend": "auto"}} From 472e3216faf05d1df1f73fcdda7045bca75968b9 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 16 Sep 2026 16:30:37 +0100 Subject: [PATCH 7/7] compose: key the NVIDIA reservation off the detected GPU _gen_gpu_deployment() emitted the "driver: nvidia" device reservation when the image name contained "cuda", "rocm" or "gpu". That is wrong in both directions: AMD and Intel hosts got a reservation for a driver they do not have, and an NVIDIA GPU served by an image whose name says neither (the vulkan-capable ramalama image) got none at all, leaving the generated compose file running on the CPU. Use the detected GPU instead, which is what the reservation describes. Signed-off-by: Oliver Walsh --- ramalama/compose.py | 28 +++++++++-- test/unit/data/test_compose/with_amd_gpu.yaml | 18 +++++++ .../data/test_compose/with_nvidia_gpu.yaml | 2 +- .../with_nvidia_gpu_selection.yaml | 25 ++++++++++ .../with_nvidia_gpu_vulkan_image.yaml | 25 ++++++++++ test/unit/test_compose.py | 47 ++++++++++++++++++- 6 files changed, 138 insertions(+), 7 deletions(-) create mode 100644 test/unit/data/test_compose/with_amd_gpu.yaml create mode 100644 test/unit/data/test_compose/with_nvidia_gpu_selection.yaml create mode 100644 test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml diff --git a/ramalama/compose.py b/ramalama/compose.py index 1c39000de..ab53e62c5 100644 --- a/ramalama/compose.py +++ b/ramalama/compose.py @@ -5,7 +5,9 @@ import shlex from typing import Optional -from ramalama.common import RAG_DIR, get_accel_env_vars, get_gpu_devices +# Live reference for checking global vars +import ramalama.common +from ramalama.common import RAG_DIR, container_cuda_visible_devices, get_accel_env_vars, get_gpu_devices from ramalama.file import PlainFile from ramalama.host_utils import format_bind_host_publish_prefix from ramalama.version import version @@ -122,6 +124,10 @@ def _gen_ports(self) -> str: def _gen_environment(self) -> str: env_vars = get_accel_env_vars() + if "CUDA_VISIBLE_DEVICES" in env_vars: + # device_ids below reserves just the selected GPUs, so the container + # renumbers them, exactly as it does for "ramalama run". + env_vars["CUDA_VISIBLE_DEVICES"] = container_cuda_visible_devices(env_vars["CUDA_VISIBLE_DEVICES"]) # Allow user to override with --env if getattr(self.args, "env", None): for e in self.args.env: @@ -137,17 +143,29 @@ def _gen_environment(self) -> str: return env_spec def _gen_gpu_deployment(self) -> str: - gpu_keywords = ["cuda", "rocm", "gpu"] - if not any(keyword in self.image.lower() for keyword in gpu_keywords): + # The "nvidia" device driver only covers NVIDIA GPUs, so key the + # reservation off the detected hardware. The image name does not + # identify it: NVIDIA can be served by the vulkan-capable ramalama + # image as well as by the cuda one. + if "CUDA_VISIBLE_DEVICES" not in get_accel_env_vars(): return "" - return """\ + # Reserve only the GPUs the user selected. CUDA_VISIBLE_DEVICES alone + # cannot narrow it: the Vulkan backend never reads that variable, so the + # other GPUs have to be kept out of the container entirely. + if selected := ramalama.common.nvidia_selected_devices: + device_ids = ", ".join(f'"{name}"' for name in selected) + reservation = f"device_ids: [{device_ids}]" + else: + reservation = "count: all" + + return f"""\ deploy: resources: reservations: devices: - driver: nvidia - count: all + {reservation} capabilities: [gpu]""" def _gen_command(self) -> str: diff --git a/test/unit/data/test_compose/with_amd_gpu.yaml b/test/unit/data/test_compose/with_amd_gpu.yaml new file mode 100644 index 000000000..e69080f6e --- /dev/null +++ b/test/unit/data/test_compose/with_amd_gpu.yaml @@ -0,0 +1,18 @@ +# Save this output to a 'docker-compose.yaml' file and run 'docker compose up'. +# +# Created with ramalama-0.1.0-test +services: + gemma-rocm: + container_name: ramalama-gemma-rocm + image: test-image/rocm:latest + volumes: + - "/models/gemma.gguf:/mnt/models/gemma.gguf:ro" + ports: + - "8080:8080" + environment: + - HIP_VISIBLE_DEVICES=0 + devices: + - "/dev/accel:/dev/accel" + - "/dev/dri:/dev/dri" + - "/dev/kfd:/dev/kfd" + restart: unless-stopped diff --git a/test/unit/data/test_compose/with_nvidia_gpu.yaml b/test/unit/data/test_compose/with_nvidia_gpu.yaml index 21f008cca..d0768ce29 100644 --- a/test/unit/data/test_compose/with_nvidia_gpu.yaml +++ b/test/unit/data/test_compose/with_nvidia_gpu.yaml @@ -10,7 +10,7 @@ services: ports: - "8080:8080" environment: - - ACCEL_ENV=true + - CUDA_VISIBLE_DEVICES=0 devices: - "/dev/accel:/dev/accel" - "/dev/dri:/dev/dri" diff --git a/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml b/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml new file mode 100644 index 000000000..e9905e863 --- /dev/null +++ b/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml @@ -0,0 +1,25 @@ +# Save this output to a 'docker-compose.yaml' file and run 'docker compose up'. +# +# Created with ramalama-0.1.0-test +services: + gemma-cuda: + container_name: ramalama-gemma-cuda + image: test-image/cuda:latest + volumes: + - "/models/gemma.gguf:/mnt/models/gemma.gguf:ro" + ports: + - "8080:8080" + environment: + - CUDA_VISIBLE_DEVICES=0 + devices: + - "/dev/accel:/dev/accel" + - "/dev/dri:/dev/dri" + - "/dev/kfd:/dev/kfd" + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["1"] + capabilities: [gpu] + restart: unless-stopped diff --git a/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml b/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml new file mode 100644 index 000000000..8c7aed896 --- /dev/null +++ b/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml @@ -0,0 +1,25 @@ +# Save this output to a 'docker-compose.yaml' file and run 'docker compose up'. +# +# Created with ramalama-0.1.0-test +services: + gemma-vulkan: + container_name: ramalama-gemma-vulkan + image: test-image/ramalama:latest + volumes: + - "/models/gemma.gguf:/mnt/models/gemma.gguf:ro" + ports: + - "8080:8080" + environment: + - CUDA_VISIBLE_DEVICES=0 + devices: + - "/dev/accel:/dev/accel" + - "/dev/dri:/dev/dri" + - "/dev/kfd:/dev/kfd" + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: all + capabilities: [gpu] + restart: unless-stopped diff --git a/test/unit/test_compose.py b/test/unit/test_compose.py index 42c32b7ef..694cad764 100644 --- a/test/unit/test_compose.py +++ b/test/unit/test_compose.py @@ -37,6 +37,8 @@ def __init__( mmproj_file_exists: bool = False, args: Args = Args(), exec_args: list = None, + accel_env_vars: dict = None, + nvidia_selected_devices: list = None, ): self.model_name = model_name self.model_src_path = model_src_path @@ -53,6 +55,8 @@ def __init__( self.mmproj_file_exists = mmproj_file_exists self.args = args self.exec_args = exec_args if exec_args is not None else [] + self.accel_env_vars = accel_env_vars if accel_env_vars is not None else {"ACCEL_ENV": "true"} + self.nvidia_selected_devices = nvidia_selected_devices or [] DATA_PATH = Path(__file__).parent / "data" / "test_compose" @@ -156,9 +160,49 @@ def __init__( model_src_path="/models/gemma.gguf", model_dest_path="/mnt/models/gemma.gguf", args=Args(image="test-image/cuda:latest"), + accel_env_vars={"CUDA_VISIBLE_DEVICES": "0"}, ), "with_nvidia_gpu.yaml", ), + ( + # The reservation follows the detected GPU, not the image name, so + # an NVIDIA GPU served by the vulkan-capable ramalama image gets one + # too. + Input( + model_name="gemma-vulkan", + model_src_path="/models/gemma.gguf", + model_dest_path="/mnt/models/gemma.gguf", + accel_env_vars={"CUDA_VISIBLE_DEVICES": "0"}, + ), + "with_nvidia_gpu_vulkan_image.yaml", + ), + ( + # A narrowed selection reserves just those GPUs, since the Vulkan + # backend cannot be filtered with CUDA_VISIBLE_DEVICES. The variable + # is renumbered to match what the container ends up seeing. + Input( + model_name="gemma-cuda", + model_src_path="/models/gemma.gguf", + model_dest_path="/mnt/models/gemma.gguf", + args=Args(image="test-image/cuda:latest"), + accel_env_vars={"CUDA_VISIBLE_DEVICES": "1"}, + nvidia_selected_devices=["1"], + ), + "with_nvidia_gpu_selection.yaml", + ), + ( + # The other direction: "nvidia" is the only driver the reservation + # can name, so an AMD GPU must not get one whatever the image is + # called. + Input( + model_name="gemma-rocm", + model_src_path="/models/gemma.gguf", + model_dest_path="/mnt/models/gemma.gguf", + args=Args(image="test-image/rocm:latest"), + accel_env_vars={"HIP_VISIBLE_DEVICES": "0"}, + ), + "with_amd_gpu.yaml", + ), ( Input( model_name="tinyllama", @@ -188,7 +232,8 @@ def test_compose_generate(input_data: Input, expected_file_name: str, monkeypatc } monkeypatch.setattr("os.path.exists", lambda path: existence.get(path, False)) - monkeypatch.setattr("ramalama.compose.get_accel_env_vars", lambda: {"ACCEL_ENV": "true"}) + monkeypatch.setattr("ramalama.compose.get_accel_env_vars", lambda: dict(input_data.accel_env_vars)) + monkeypatch.setattr("ramalama.common.nvidia_selected_devices", list(input_data.nvidia_selected_devices)) monkeypatch.setattr("ramalama.compose.version", lambda: "0.1.0-test") draft_model_paths = (None, None)