From 6bbe60d6062af2ac6b91d8acf77c7288041132a3 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 16 Sep 2026 16:37:13 +0100 Subject: [PATCH 1/3] llama.cpp: fix image selection when auto resolves to another backend Image selection keys off the resolved backend's GPU environment variable, which in auto mode is not the one the hardware was detected as. Two things fall out of that for a GPU that resolves to the vulkan backend: An image the user configured for their GPU, e.g. images.HIP_VISIBLE_DEVICES, is never looked up, because the lookup happens after the key has been rewritten to GGML_VK_VISIBLE_DEVICES. Consult the detected GPU first in auto mode, so a pin for the hardware wins over one for the resolved backend. default_image / RAMALAMA_DEFAULT_IMAGE is ignored, because the vulkan entry in the image table hardcodes the published quay.io/ramalama/ramalama tag. With no GPU at all the same code path returns default_image, so the two disagreed. Drop the entry and let it fall through. Signed-off-by: Oliver Walsh --- .../plugins/runtimes/inference/llama_cpp.py | 15 ++++++++-- test/unit/test_inference_engine_plugins.py | 30 +++++++++++++++++++ 2 files changed, 42 insertions(+), 3 deletions(-) diff --git a/ramalama/plugins/runtimes/inference/llama_cpp.py b/ramalama/plugins/runtimes/inference/llama_cpp.py index 0a4a537e6..4e42263cb 100644 --- a/ramalama/plugins/runtimes/inference/llama_cpp.py +++ b/ramalama/plugins/runtimes/inference/llama_cpp.py @@ -206,11 +206,13 @@ def get_available_backends() -> list[str]: return ["auto", "vulkan"] +# GGML_VK_VISIBLE_DEVICES, the vulkan backend, is deliberately absent: it runs +# in the default image, and hardcoding the published tag here would ignore +# default_image / RAMALAMA_DEFAULT_IMAGE whenever a GPU is detected. _LLAMA_CPP_IMAGES: dict[str, str] = { "ASAHI_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/asahi"), "ASCEND_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/cann"), "CUDA_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/cuda"), - "GGML_VK_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/ramalama"), "HIP_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/rocm"), "INTEL_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/intel-gpu"), "OPENVINO_VISIBLE_DEVICES": version_tagged_image("quay.io/ramalama/openvino"), @@ -374,8 +376,15 @@ def get_container_image(self, config: Any, detected_gpu_type: str) -> Optional[s if gpu_type == "GGML_VK_VISIBLE_DEVICES" and detected_gpu_type == "CUDA_VISIBLE_DEVICES": warn_without_nvidia_vulkan_icd() - override = config.images.get(gpu_type) if gpu_type else None - return override if override else _LLAMA_CPP_IMAGES.get(gpu_type, config.default_image) + # In auto mode an image configured for the detected GPU is the user + # pinning an image for their hardware, so it takes precedence over one + # configured for the backend that was resolved from that hardware. + override_keys = [detected_gpu_type, gpu_type] if backend == "auto" else [gpu_type] + for key in override_keys: + if key and (override := config.images.get(key)): + return override + + return _LLAMA_CPP_IMAGES.get(gpu_type, config.default_image) def _container_image_is_ggml(self, args: argparse.Namespace) -> bool: if not args.container or args.dryrun: diff --git a/test/unit/test_inference_engine_plugins.py b/test/unit/test_inference_engine_plugins.py index 0a5198a54..ee56aea9c 100644 --- a/test/unit/test_inference_engine_plugins.py +++ b/test/unit/test_inference_engine_plugins.py @@ -752,6 +752,36 @@ def test_get_container_image_user_override(self): image = self.plugin.get_container_image(config, "CUDA_VISIBLE_DEVICES") assert image == "custom/cuda:v1.0" + def test_get_container_image_vulkan_follows_default_image(self): + # AMD resolves to the vulkan backend, which runs in the default image, + # so RAMALAMA_DEFAULT_IMAGE has to be honoured here too. + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "auto"}} + config.images.get.return_value = None + config.default_image = "my-registry.example/ramalama:snapshot" + image = self.plugin.get_container_image(config, "HIP_VISIBLE_DEVICES") + assert image == "my-registry.example/ramalama:snapshot" + + def test_get_container_image_user_override_for_detected_gpu(self): + # The override is keyed by the detected GPU, while auto resolves to the + # vulkan backend; the image the user pinned for their hardware wins. + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "auto"}} + config.images.get.side_effect = lambda key, default=None: { + "HIP_VISIBLE_DEVICES": "custom/rocm:v1.0", + }.get(key, default) + image = self.plugin.get_container_image(config, "HIP_VISIBLE_DEVICES") + assert image == "custom/rocm:v1.0" + + def test_get_container_image_user_override_for_resolved_backend(self): + config = MagicMock() + config.runtimes = {"llama_cpp": {"backend": "auto"}} + config.images.get.side_effect = lambda key, default=None: { + "GGML_VK_VISIBLE_DEVICES": "custom/vulkan:v1.0", + }.get(key, default) + image = self.plugin.get_container_image(config, "HIP_VISIBLE_DEVICES") + assert image == "custom/vulkan:v1.0" + @patch.dict("os.environ", clear=True, INTEL_VISIBLE_DEVICES="1") def test_intel_no_openvino(self): ns = make_ns() From 839deec455b5816e7eae0f8e55b2238348bb8fbb Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Wed, 9 Sep 2026 17:26:17 +0100 Subject: [PATCH 2/3] common: detect WSL2 when choosing GPU backend defaults Backend preferences already avoided Vulkan on Windows, because WSL2 exposes GPUs through /dev/dxg and Vulkan there means mesa's dzn driver translating to D3D12 (no cooperative matrix support, no compute tuning) or a silent llvmpipe fallback. That check used platform.system() == "Windows", which is only true for a native Windows interpreter. Running ramalama inside the WSL2 distro itself reports "Linux" and got the Linux defaults. Add is_windows_or_wsl() alongside the other platform predicates in common, covering both cases, and use it for the AMD and Intel preference overrides. The /dev/dxg passthrough and /usr/lib/wsl bind mount in add_device_options() move to the same predicate. They are needed wherever the GPU comes in through WSL, so leaving them on the Windows check would have picked the sycl image inside a distro and then handed it no GPU at all. Describe the split in the docs as WSL2 rather than Windows to match, and reflow the dangling comment in the sample config while touching it. in_wsl() caches its /proc read, so clear it around every test in conftest, as in_toolbox already does. Signed-off-by: Oliver Walsh --- docs/options/backend.md | 9 ++-- docs/ramalama-bench.1.md | 9 ++-- docs/ramalama-perplexity.1.md | 9 ++-- docs/ramalama-run.1.md | 9 ++-- docs/ramalama-sandbox-goose.1.md | 9 ++-- docs/ramalama-sandbox-opencode.1.md | 9 ++-- docs/ramalama-sandbox-pi.1.md | 9 ++-- docs/ramalama-serve.1.md | 9 ++-- docs/ramalama.conf | 11 ++--- docs/ramalama.conf.5.md | 6 +-- ramalama/common.py | 27 +++++++++++ ramalama/engine.py | 16 ++++--- .../plugins/runtimes/inference/llama_cpp.py | 8 ++-- test/unit/conftest.py | 11 +++++ test/unit/test_common.py | 45 +++++++++++++++++++ test/unit/test_engine.py | 39 +++++++++++++++- test/unit/test_inference_engine_plugins.py | 27 ++++++----- 17 files changed, 208 insertions(+), 54 deletions(-) diff --git a/docs/options/backend.md b/docs/options/backend.md index 41d6a40da..0ce3ade46 100644 --- a/docs/options/backend.md +++ b/docs/options/backend.md @@ -9,16 +9,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-bench.1.md b/docs/ramalama-bench.1.md index f82372170..c7bd8d3fe 100644 --- a/docs/ramalama-bench.1.md +++ b/docs/ramalama-bench.1.md @@ -45,16 +45,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-perplexity.1.md b/docs/ramalama-perplexity.1.md index 6cff430ba..c89c3708d 100644 --- a/docs/ramalama-perplexity.1.md +++ b/docs/ramalama-perplexity.1.md @@ -45,16 +45,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-run.1.md b/docs/ramalama-run.1.md index 5ed45f709..1022faf66 100644 --- a/docs/ramalama-run.1.md +++ b/docs/ramalama-run.1.md @@ -57,16 +57,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-sandbox-goose.1.md b/docs/ramalama-sandbox-goose.1.md index 35c1fd3af..f7e52a404 100644 --- a/docs/ramalama-sandbox-goose.1.md +++ b/docs/ramalama-sandbox-goose.1.md @@ -53,16 +53,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-sandbox-opencode.1.md b/docs/ramalama-sandbox-opencode.1.md index 5af6bf194..b067e8679 100644 --- a/docs/ramalama-sandbox-opencode.1.md +++ b/docs/ramalama-sandbox-opencode.1.md @@ -53,16 +53,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-sandbox-pi.1.md b/docs/ramalama-sandbox-pi.1.md index 868af0d3e..44129449f 100644 --- a/docs/ramalama-sandbox-pi.1.md +++ b/docs/ramalama-sandbox-pi.1.md @@ -57,16 +57,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama-serve.1.md b/docs/ramalama-serve.1.md index 83400ba02..20e85951c 100644 --- a/docs/ramalama-serve.1.md +++ b/docs/ramalama-serve.1.md @@ -86,16 +86,19 @@ GPU backend to use for inference (default: auto). Available backends depend on the detected GPU hardware. **auto** (default): Automatically selects the preferred backend based on your GPU: -- **AMD GPUs**: vulkan (Linux/macOS) or rocm (Windows) +- **AMD GPUs**: vulkan (Linux/macOS) or rocm (WSL2) - **NVIDIA GPUs**: cuda; vulkan available as explicit option -- **Intel GPUs**: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +- **Intel GPUs**: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - **Ascend NPUs**: cann - **MUSA GPUs**: musa - **No GPU**: vulkan (CPU fallback) **Platform-specific behavior**: - On **Linux/macOS**, Vulkan provides broad compatibility and is preferred for AMD and Intel GPUs -- On **Windows**, vulkan is not supported on WSL2, so vendor-specific backends (rocm, sycl) are preferred +- On **WSL2**, Vulkan is a poor default, so vendor-specific backends (rocm, sycl) are preferred. It + remains available, and `--backend=vulkan` still selects it. This covers both Windows, where + containers run in the WSL2-backed machine, and ramalama running inside a WSL2 distro, which + otherwise looks like Linux **Explicit backend selection**: - **vulkan**: Use Vulkan-based inference (compatible with AMD, NVIDIA, Intel, and CPU) diff --git a/docs/ramalama.conf b/docs/ramalama.conf index cc54ea239..4b24d6d00 100644 --- a/docs/ramalama.conf +++ b/docs/ramalama.conf @@ -240,9 +240,9 @@ # # Valid options: auto, vulkan, rocm, cuda, sycl, openvino, cann, musa # - auto (default): Automatically selects the preferred backend based on detected GPU -# - AMD GPUs: vulkan (Linux/macOS) or rocm (Windows) +# - AMD GPUs: vulkan (Linux/macOS) or rocm (WSL2) # - NVIDIA GPUs: cuda; vulkan available as explicit option -# - Intel GPUs: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option +# - Intel GPUs: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option # - Ascend NPUs: cann # - MUSA GPUs: musa # - No GPU: vulkan (CPU fallback) @@ -254,9 +254,10 @@ # - cann: Use Huawei CANN backend (Ascend NPUs only); uses quay.io/ramalama/cann # - musa: Use Moore Threads MUSA backend (MUSA GPUs only); uses quay.io/ramalama/musa # -# Platform-specific behavior: On Windows, vulkan is not supported on WSL2, so -# vendor-specific backends (rocm for AMD, sycl for Intel) are automatically -# preferred when using backend="auto". +# Platform-specific behavior: vulkan is not supported on WSL2, so vendor-specific +# backends (rocm for AMD, sycl for Intel) are automatically preferred when using +# backend="auto". This covers both Windows and ramalama running inside a WSL2 +# distro, which otherwise looks like Linux. # #backend = "auto" diff --git a/docs/ramalama.conf.5.md b/docs/ramalama.conf.5.md index 529913a09..261d76eb6 100644 --- a/docs/ramalama.conf.5.md +++ b/docs/ramalama.conf.5.md @@ -239,9 +239,9 @@ This setting affects which container image is selected and how GPU resources are Valid options: `auto`, `vulkan`, `rocm`, `cuda`, `sycl`, `openvino`, `cann`, `musa`. - **auto** (default): Automatically selects the preferred backend based on detected GPU: - - AMD GPUs: vulkan (Linux/macOS) or rocm (Windows) + - AMD GPUs: vulkan (Linux/macOS) or rocm (WSL2) - NVIDIA GPUs: cuda; vulkan available as explicit option - - Intel GPUs: vulkan (Linux/macOS) or sycl (Windows); openvino available as explicit option + - Intel GPUs: vulkan (Linux/macOS) or sycl (WSL2); openvino available as explicit option - Ascend NPUs: cann - MUSA GPUs: musa - No GPU: vulkan (CPU fallback) @@ -254,7 +254,7 @@ Valid options: `auto`, `vulkan`, `rocm`, `cuda`, `sycl`, `openvino`, `cann`, `mu - **cann**: Use Huawei CANN backend (Ascend NPUs only); uses `quay.io/ramalama/cann` - **musa**: Use Moore Threads MUSA backend (MUSA GPUs only); uses `quay.io/ramalama/musa` -**Platform-specific behavior**: On Windows, vulkan is not supported on WSL2, so vendor-specific backends (rocm for AMD, sycl for Intel) are automatically preferred when using `backend="auto"`. +**Platform-specific behavior**: vulkan is not supported on WSL2, so vendor-specific backends (rocm for AMD, sycl for Intel) are automatically preferred when using `backend="auto"`. This covers both Windows and ramalama running inside a WSL2 distro, which otherwise looks like Linux. Example configuration: diff --git a/ramalama/common.py b/ramalama/common.py index fa38f08bf..701a3f838 100644 --- a/ramalama/common.py +++ b/ramalama/common.py @@ -488,6 +488,33 @@ def check_metal(args: ContainerArgType) -> bool: return platform.system() == "Darwin" +@lru_cache(maxsize=1) +def in_wsl() -> bool: + """True when the interpreter itself is running inside a WSL distro. + + False on native Windows, where ramalama drives a podman machine instead. + """ + try: + with open("/proc/sys/kernel/osrelease") as f: + return "microsoft" in f.read().lower() + except OSError: + return False + + +def is_windows_or_wsl() -> bool: + """True where containers reach GPUs through WSL rather than native devices. + + Covers both a native Windows interpreter, which runs containers in the + WSL2-backed podman machine, and ramalama running inside a WSL distro + itself, which platform.system() reports as "Linux". + + WSL exposes GPUs through /dev/dxg, so Vulkan there means mesa's dzn driver + translating to D3D12 (no cooperative matrix support, no compute tuning) or + a silent llvmpipe fallback. + """ + return platform.system() == "Windows" or in_wsl() + + @lru_cache(maxsize=1) def has_nvidia_vulkan_icd() -> bool: """True when NVIDIA's Vulkan ICD manifest is installed on the host. diff --git a/ramalama/engine.py b/ramalama/engine.py index 01ffdadc1..cfffa3e7c 100644 --- a/ramalama/engine.py +++ b/ramalama/engine.py @@ -3,7 +3,6 @@ import glob import json import os -import platform import subprocess import sys import time @@ -23,6 +22,7 @@ genname, get_accel_env_vars, host_path, + is_windows_or_wsl, perror, run_cmd, ) @@ -124,7 +124,7 @@ def add_device_options(self): for dev in glob.glob(path): self.exec_args += ["--device", dev] - intel_windows_added = False + wsl_devices_added = False for k, v in get_accel_env_vars().items(): # Special case for Cuda if k == "CUDA_VISIBLE_DEVICES": @@ -147,11 +147,17 @@ def add_device_options(self): v = container_cuda_visible_devices(v) elif k == "MUSA_VISIBLE_DEVICES": self.exec_args += ["--env", "MTHREADS_VISIBLE_DEVICES=all"] - elif k == "INTEL_VISIBLE_DEVICES": - if platform.system() == "Windows" and not intel_windows_added: + elif k in ("HIP_VISIBLE_DEVICES", "INTEL_VISIBLE_DEVICES"): + # WSL exposes the GPU as /dev/dxg with its driver libraries in + # /usr/lib/wsl, whether ramalama runs on native Windows against + # the podman machine or inside the distro itself, where + # platform.system() reports "Linux". That is how both the AMD + # and the Intel GPU come in, so the rocm backend needs it as + # much as sycl does. + if is_windows_or_wsl() and not wsl_devices_added: self.exec_args += ["--device", "/dev/dxg"] self.exec_args += ["--mount", "type=bind,src=/usr/lib/wsl,dst=/usr/lib/wsl"] - intel_windows_added = True + wsl_devices_added = True self.exec_args += ["-e", f"{k}={v}"] diff --git a/ramalama/plugins/runtimes/inference/llama_cpp.py b/ramalama/plugins/runtimes/inference/llama_cpp.py index 4e42263cb..4ddc598a3 100644 --- a/ramalama/plugins/runtimes/inference/llama_cpp.py +++ b/ramalama/plugins/runtimes/inference/llama_cpp.py @@ -4,7 +4,6 @@ import copy import json import os -import platform import shutil import subprocess import sys @@ -42,6 +41,7 @@ genname, get_gpu_type_env_vars, has_nvidia_vulkan_icd, + is_windows_or_wsl, run_cmd, set_accel_env_vars, set_gpu_type_env_vars, @@ -141,9 +141,7 @@ def __call__(self, parser, namespace, values, option_string=None): def get_gpu_backend_preferences(gpu_type: str) -> list[str]: """Returns preferred backends for a given GPU type in order of preference. - On Windows, vulkan is not supported on WSL2, so vendor backends are preferred.""" - is_windows = platform.system() == "Windows" - + Vulkan is a poor default on WSL2, so vendor backends are preferred there.""" preferences = { "HIP_VISIBLE_DEVICES": ["vulkan", "rocm"], # AMD: Vulkan preferred "CUDA_VISIBLE_DEVICES": ["cuda", "vulkan"], # NVIDIA: CUDA preferred @@ -154,7 +152,7 @@ def get_gpu_backend_preferences(gpu_type: str) -> list[str]: "GGML_VK_VISIBLE_DEVICES": ["vulkan"], # Vulkan: Vulkan only } - if is_windows: + if is_windows_or_wsl(): preferences["HIP_VISIBLE_DEVICES"] = ["rocm", "vulkan"] preferences["INTEL_VISIBLE_DEVICES"] = ["sycl", "vulkan", "openvino"] diff --git a/test/unit/conftest.py b/test/unit/conftest.py index 50d6bb225..49c5b7faa 100644 --- a/test/unit/conftest.py +++ b/test/unit/conftest.py @@ -45,6 +45,17 @@ def _isolate_from_toolbox(): in_toolbox.cache_clear() +@pytest.fixture(autouse=True) +def _clear_wsl_cache(): + """in_wsl() caches its /proc read, so tests patching it must not leak into + each other or inherit whatever the host answered first.""" + from ramalama.common import in_wsl + + in_wsl.cache_clear() + yield + in_wsl.cache_clear() + + @pytest.fixture(autouse=True) def _clear_vulkan_icd_cache(): """The ICD probe and the warning it drives are cached for the process, so diff --git a/test/unit/test_common.py b/test/unit/test_common.py index cd074e1b5..168e9de79 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -31,6 +31,8 @@ host_cmd, host_path, in_toolbox, + in_wsl, + is_windows_or_wsl, load_cdi_config, populate_volume_from_image, rm_until_substring, @@ -964,3 +966,46 @@ def test_icd_installed(self, icd_dir): def test_only_other_vendors(self): with patch("ramalama.common.glob.glob", return_value=[]): assert not has_nvidia_vulkan_icd() + + +class TestIsWindowsOrWsl: + """is_windows_or_wsl() must catch both native Windows and ramalama running inside a WSL2 distro.""" + + @pytest.mark.parametrize( + "osrelease,expected", + [ + ("5.15.167.4-microsoft-standard-WSL2\n", True), + ("6.6.87.2-microsoft-standard-WSL2+\n", True), + ("7.1.13-100.fc43.x86_64\n", False), + ], + ) + def test_in_wsl_reads_osrelease(self, osrelease, expected): + with patch("builtins.open", mock_open(read_data=osrelease)): + assert in_wsl() == expected + + def test_in_wsl_missing_osrelease(self): + with patch("builtins.open", side_effect=OSError("no /proc")): + assert not in_wsl() + + def test_native_windows(self): + # No /proc to read on a native Windows interpreter. + with ( + patch("ramalama.common.platform.system", return_value="Windows"), + patch("builtins.open", side_effect=OSError("no /proc")), + ): + assert is_windows_or_wsl() + + def test_inside_wsl_distro(self): + # platform.system() reports "Linux" from inside the distro. + with ( + patch("ramalama.common.platform.system", return_value="Linux"), + patch("builtins.open", mock_open(read_data="5.15.167.4-microsoft-standard-WSL2\n")), + ): + assert is_windows_or_wsl() + + def test_native_linux(self): + with ( + patch("ramalama.common.platform.system", return_value="Linux"), + patch("builtins.open", mock_open(read_data="7.1.13-100.fc43.x86_64\n")), + ): + assert not is_windows_or_wsl() diff --git a/test/unit/test_engine.py b/test/unit/test_engine.py index 7f2b0c78e..e619e3bfd 100644 --- a/test/unit/test_engine.py +++ b/test/unit/test_engine.py @@ -58,6 +58,41 @@ def test_cuda_device_options_docker_selection(self): exec_args = self._cuda_device_args("docker", "1,2", selected=["1", "2"]) self.assertEqual(exec_args, ["--gpus", '"device=1,2"', "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + def _wsl_device_args(self, accel_env_var, windows_or_wsl): + engine = ramalama.engine.Engine(self.base_args) + engine.exec_args = [] + with ( + patch("ramalama.engine.get_accel_env_vars", return_value={accel_env_var: "0"}), + patch("ramalama.engine.is_windows_or_wsl", return_value=windows_or_wsl), + patch("glob.glob", return_value=[]), + patch.object(ramalama.common, "podman_machine_accel", False), + ): + engine.add_device_options() + return engine.exec_args + + def test_intel_device_options_native(self): + exec_args = self._wsl_device_args("INTEL_VISIBLE_DEVICES", False) + self.assertEqual(exec_args, ["-e", "INTEL_VISIBLE_DEVICES=0"]) + + def test_wsl_device_options(self): + # WSL exposes the AMD GPU the same way as the Intel one. Also covers + # ramalama running inside a WSL distro, where platform.system() reports + # "Linux" but the GPU is still /dev/dxg. + for accel_env_var in ("HIP_VISIBLE_DEVICES", "INTEL_VISIBLE_DEVICES"): + with self.subTest(accel_env_var=accel_env_var): + exec_args = self._wsl_device_args(accel_env_var, True) + self.assertEqual( + exec_args, + [ + "--device", + "/dev/dxg", + "--mount", + "type=bind,src=/usr/lib/wsl,dst=/usr/lib/wsl", + "-e", + f"{accel_env_var}=0", + ], + ) + def test_add_container_labels(self): args = Namespace(**vars(self.base_args), MODEL="test-model", port="8080", subcommand="run") engine = ramalama.engine.Engine(args) @@ -249,7 +284,7 @@ def test_remove_network_dryrun_skips_remove(self, mock_run_cmd): "none-host", ], ) -@patch("ramalama.engine.platform.system", return_value="Linux") +@patch("ramalama.host_utils.platform.system", return_value="Linux") def test_add_port_with_host(mock_system, host, port, expected_port_arg): base_args = Namespace( engine="podman", @@ -294,7 +329,7 @@ def test_add_port_with_host_on_vm_engine(system, host, port, expected_port_arg): host=host, port=port, ) - with patch("ramalama.engine.platform.system", return_value=system): + with patch("ramalama.host_utils.platform.system", return_value=system): engine = ramalama.engine.Engine(base_args) p_index = engine.exec_args.index("-p") assert engine.exec_args[p_index + 1] == expected_port_arg diff --git a/test/unit/test_inference_engine_plugins.py b/test/unit/test_inference_engine_plugins.py index ee56aea9c..80f45ad29 100644 --- a/test/unit/test_inference_engine_plugins.py +++ b/test/unit/test_inference_engine_plugins.py @@ -29,6 +29,13 @@ from ramalama.plugins.runtimes.inference.vllm import VllmPlugin +@pytest.fixture(autouse=True) +def not_windows_or_wsl(monkeypatch): + """Backend defaults differ on Windows and WSL, so pin the native path by + default. Tests covering that behaviour override this.""" + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.is_windows_or_wsl", lambda: False) + + def make_ns( container=True, file=None, @@ -1326,8 +1333,8 @@ def test_backend_selection(backend: str, gpu_env: str, expected_result: str, mon @pytest.mark.parametrize( "backend,gpu_env,expected_result", [ - # Auto mode on Windows: ROCm for AMD, CUDA for NVIDIA, sycl for Intel - ("auto", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), # AMD -> ROCm on Windows + # Auto mode on WSL2: ROCm for AMD, CUDA for NVIDIA, sycl for Intel + ("auto", "HIP_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/rocm")), # AMD -> ROCm on WSL2 ("auto", "CUDA_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/cuda")), # NVIDIA -> CUDA ("auto", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/intel-gpu")), # Intel -> sycl # Explicit backends still work, vulkan included @@ -1340,10 +1347,10 @@ def test_backend_selection(backend: str, gpu_env: str, expected_result: str, mon ("openvino", "INTEL_VISIBLE_DEVICES", version_tagged_image("quay.io/ramalama/openvino")), ], ) -def test_backend_selection_windows(backend: str, gpu_env: str, expected_result: str, monkeypatch): - """Test that Windows defaults to vendor-specific backends for AMD and Intel.""" +def test_backend_selection_wsl(backend: str, gpu_env: str, expected_result: str, monkeypatch): + """Test that WSL2 defaults to vendor-specific backends for AMD and Intel.""" monkeypatch.setattr("ramalama.common.get_accel", lambda: "none") - monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.platform.system", lambda: "Windows") + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.is_windows_or_wsl", lambda: True) with NamedTemporaryFile('w', delete_on_close=False) as f: f.write(f"""\ @@ -1475,16 +1482,16 @@ def test_get_available_backends(gpu_env: Optional[str], expected_backends: list[ @pytest.mark.parametrize( "gpu_env,expected_backends", [ - ("HIP_VISIBLE_DEVICES", ["auto", "rocm", "vulkan"]), # AMD: ROCm preferred on Windows + ("HIP_VISIBLE_DEVICES", ["auto", "rocm", "vulkan"]), # AMD: ROCm preferred on WSL2 ("CUDA_VISIBLE_DEVICES", ["auto", "cuda", "vulkan"]), # NVIDIA: same on all platforms - ("INTEL_VISIBLE_DEVICES", ["auto", "sycl", "vulkan", "openvino"]), # Intel: sycl preferred on Windows + ("INTEL_VISIBLE_DEVICES", ["auto", "sycl", "vulkan", "openvino"]), # Intel: sycl preferred on WSL2 (None, ["auto", "vulkan"]), # No GPU: same on all platforms ], ) -def test_get_available_backends_windows(gpu_env: Optional[str], expected_backends: list[str], monkeypatch): - """Test that available backends on Windows prefer vendor-specific backends.""" +def test_get_available_backends_wsl(gpu_env: Optional[str], expected_backends: list[str], monkeypatch): + """Test that available backends on WSL2 prefer vendor-specific backends.""" monkeypatch.setattr("ramalama.common.get_accel", lambda: "none") - monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.platform.system", lambda: "Windows") + monkeypatch.setattr("ramalama.plugins.runtimes.inference.llama_cpp.is_windows_or_wsl", lambda: True) env = {} if gpu_env: From 4dee9cc802146ca63f8afb14456e80bf321da071 Mon Sep 17 00:00:00 2001 From: Oliver Walsh Date: Thu, 17 Sep 2026 16:56:03 +0100 Subject: [PATCH 3/3] GPU: keep the host's other GPUs out of NVIDIA containers The NVIDIA container toolkit injects the device nodes of the GPUs that were asked for, DRM nodes included. Mapping the host's GPU devices in on top of that - /dev/dri, /dev/kfd, /dev/accel - can only add devices that are not the accelerator in play: an iGPU on a hybrid host, or a GPU left out of a narrowed CUDA_VISIBLE_DEVICES selection. That was harmless while llama.cpp ran CUDA, which ignores any device it was not given, but the Vulkan backend offloads onto every device it can enumerate. The extra GPU is likely to be far slower than the one that was asked for, and on a narrowed selection it is one the user asked to keep out. So hand the host's GPU devices over only when NVIDIA is not the accelerator: in get_gpu_devices() for "--generate compose" and "--generate kube", mirrored in engine.py for run and serve and in quadlet.py for "--generate quadlet". "--device /dev/dri" still passes them in for anyone who wants them. Signed-off-by: Oliver Walsh --- docs/ramalama-cuda.7.md | 2 + ramalama/common.py | 16 +++++++- ramalama/compose.py | 2 +- ramalama/engine.py | 7 +++- ramalama/kube.py | 4 +- ramalama/quadlet.py | 28 ++++++++++--- .../data/test_compose/with_nvidia_gpu.yaml | 4 -- .../with_nvidia_gpu_selection.yaml | 4 -- .../with_nvidia_gpu_vulkan_image.yaml | 4 -- .../test_quadlet/basic/tinyllama.container | 3 -- .../draft_model/tinyllama.container | 3 -- .../test_quadlet/empty/tinyllama.container | 3 -- .../modelfromstore/modelfromstore.container | 3 -- .../modelfromstore_add_to_unit.container | 3 -- .../modelfromstore_ct.container | 3 -- .../modelfromstore_mmproj.container | 3 -- .../multipart/gpt-oss-120b.container | 3 -- .../oci_basic/oci-model.container | 3 -- .../oci_port/oci-model-port.container | 3 -- .../oci_rag/oci-model-rag.container | 3 -- .../portmapping/tinyllama.container | 3 -- test/unit/test_common.py | 29 ++++++++++++++ test/unit/test_engine.py | 39 +++++++++++++++++++ test/unit/test_quadlet.py | 22 +++++++++++ 24 files changed, 137 insertions(+), 60 deletions(-) diff --git a/docs/ramalama-cuda.7.md b/docs/ramalama-cuda.7.md index b804dfb82..bee71c0ec 100644 --- a/docs/ramalama-cuda.7.md +++ b/docs/ramalama-cuda.7.md @@ -138,6 +138,8 @@ ramalama run granite This is particularly useful in multi-GPU systems where you want to dedicate specific GPUs to different workloads. +Where the CDI configuration has an entry for each GPU, only the selected ones are passed into the container by the NVIDIA container toolkit; where it only defines `all`, every detected GPU is. Either way the host's other GPU devices, such as `/dev/dri` for an integrated GPU, are left out, since the Vulkan backend offloads onto every device it can enumerate and would otherwise use them. Pass `--device /dev/dri` to `ramalama run` or `ramalama serve` to add them back. + If `CUDA_VISIBLE_DEVICES` is set to an empty string, RamaLama treats it as unset and follows the default GPU-selection behavior. ```bash diff --git a/ramalama/common.py b/ramalama/common.py index 701a3f838..de5e690ea 100644 --- a/ramalama/common.py +++ b/ramalama/common.py @@ -11,7 +11,7 @@ import string import subprocess import sys -from collections.abc import Callable, Sequence +from collections.abc import Callable, Collection, Sequence from dataclasses import dataclass from functools import lru_cache from pathlib import Path @@ -749,7 +749,19 @@ def set_gpu_type_env_vars(): ] -def get_gpu_devices(): +def get_gpu_devices(accel_env_vars: Optional[Collection[str]] = None) -> dict[str, str]: + """The host GPU devices to hand to the container, given the accelerator in play. + + An NVIDIA GPU does not come in this way: the container toolkit injects the + device nodes of the GPUs that were asked for, DRM nodes included. Mapping + the host's GPU devices in as well can then only add ones that are not the + accelerator in play - an iGPU on a hybrid host, or a GPU left out of a + narrowed selection - and llama.cpp's Vulkan backend offloads onto every + device it can enumerate. "--device" remains for anyone who wants them. + """ + if "CUDA_VISIBLE_DEVICES" in (get_gpu_type_env_vars() if accel_env_vars is None else accel_env_vars): + return {} + devices = {} for dev in ["dri", "kfd", "accel"]: path = "/dev/" + dev diff --git a/ramalama/compose.py b/ramalama/compose.py index ab53e62c5..449b3b132 100644 --- a/ramalama/compose.py +++ b/ramalama/compose.py @@ -98,7 +98,7 @@ def _gen_mmproj_volume(self) -> str: return f'\n - "{self.src_mmproj_path}:{self.dest_mmproj_path}:ro"' def _gen_devices(self) -> str: - devices = get_gpu_devices() + devices = get_gpu_devices(get_accel_env_vars()) if not devices: return "" diff --git a/ramalama/engine.py b/ramalama/engine.py index cfffa3e7c..a6918a19a 100644 --- a/ramalama/engine.py +++ b/ramalama/engine.py @@ -21,6 +21,7 @@ exec_cmd, genname, get_accel_env_vars, + get_gpu_devices, host_path, is_windows_or_wsl, perror, @@ -120,12 +121,14 @@ def add_device_options(self): if ramalama.common.podman_machine_accel: self.exec_args += ["--device", "/dev/dri"] - for path in ["/dev/dri", "/dev/kfd", "/dev/accel", "/dev/davinci*", "/dev/devmm_svm", "/dev/hisi_hdc"]: + env_vars = get_accel_env_vars() + gpu_devices = list(get_gpu_devices(env_vars).values()) + for path in [*gpu_devices, "/dev/davinci*", "/dev/devmm_svm", "/dev/hisi_hdc"]: for dev in glob.glob(path): self.exec_args += ["--device", dev] wsl_devices_added = False - for k, v in get_accel_env_vars().items(): + for k, v in env_vars.items(): # Special case for Cuda if k == "CUDA_VISIBLE_DEVICES": # Pass in only the GPUs the user selected rather than all of diff --git a/ramalama/kube.py b/ramalama/kube.py index a8c4ca3be..443d5a78f 100644 --- a/ramalama/kube.py +++ b/ramalama/kube.py @@ -85,7 +85,7 @@ def _gen_volumes(self) -> Tuple[str, str]: def _gen_devices(self) -> Tuple[str, str]: mounts = "" volumes = "" - for name, path in get_gpu_devices().items(): + for name, path in get_gpu_devices(get_accel_env_vars()).items(): mounts += f""" - mountPath: {path} name: {name}""" @@ -219,7 +219,7 @@ def __gen_resources(self) -> str: limits: 'nvidia.com/gpu=all': 1""" - devices = get_gpu_devices() + devices = get_gpu_devices(get_accel_env_vars()) if devices: limits = "".join(f"\n 'podman.io/device={path}': 1" for path in devices.values()) return f""" diff --git a/ramalama/quadlet.py b/ramalama/quadlet.py index 88564a347..a5b3e5382 100644 --- a/ramalama/quadlet.py +++ b/ramalama/quadlet.py @@ -4,7 +4,16 @@ import shlex from typing import Optional, Tuple -from ramalama.common import MNT_DIR, RAG_DIR, ContainerEntryPoint, get_accel, get_accel_env_vars +# Live reference for checking global vars +import ramalama.common +from ramalama.common import ( + MNT_DIR, + RAG_DIR, + ContainerEntryPoint, + container_cuda_visible_devices, + get_accel, + get_accel_env_vars, +) from ramalama.file import UnitFile from ramalama.host_utils import format_bind_host_publish_prefix, is_loopback_bind_host @@ -68,11 +77,16 @@ def generate(self) -> list[UnitFile]: quadlet_file = UnitFile(container_file_name) quadlet_file.add("Unit", "Description", f"RamaLama {self.name} AI Model Service") quadlet_file.add("Unit", "After", "local-fs.target") - quadlet_file.add("Container", "AddDevice", "-/dev/accel") - quadlet_file.add("Container", "AddDevice", "-/dev/dri") - quadlet_file.add("Container", "AddDevice", "-/dev/kfd") if get_accel() == "cuda": - quadlet_file.add("Container", "AddDevice", "nvidia.com/gpu=all") + # The container toolkit brings in the NVIDIA device nodes itself, so + # the host's GPU devices would only add ones that are not the GPU in + # play, such as an iGPU. See get_gpu_devices(). + for name in ramalama.common.nvidia_selected_devices or ["all"]: + quadlet_file.add("Container", "AddDevice", f"nvidia.com/gpu={name}") + else: + quadlet_file.add("Container", "AddDevice", "-/dev/accel") + quadlet_file.add("Container", "AddDevice", "-/dev/dri") + quadlet_file.add("Container", "AddDevice", "-/dev/kfd") quadlet_file.add("Container", "Image", f"{self.image}") quadlet_file.add("Container", "RunInit", "true") quadlet_file.add("Container", "Environment", "HOME=/tmp") @@ -132,6 +146,10 @@ def _gen_mmproj_volume(self, quadlet_file: UnitFile): def _gen_env(self, quadlet_file: UnitFile): env_var_string = "" for k, v in get_accel_env_vars().items(): + if k == "CUDA_VISIBLE_DEVICES": + # AddDevice above passes in just the selected GPUs, so the + # container renumbers them, exactly as it does for "ramalama run". + v = container_cuda_visible_devices(v) quadlet_file.add("Container", "Environment", f"{k}={v}") for e in self.args.env: quadlet_file.add("Container", "Environment", f"{e}") diff --git a/test/unit/data/test_compose/with_nvidia_gpu.yaml b/test/unit/data/test_compose/with_nvidia_gpu.yaml index d0768ce29..36d06108f 100644 --- a/test/unit/data/test_compose/with_nvidia_gpu.yaml +++ b/test/unit/data/test_compose/with_nvidia_gpu.yaml @@ -11,10 +11,6 @@ services: - "8080:8080" environment: - CUDA_VISIBLE_DEVICES=0 - devices: - - "/dev/accel:/dev/accel" - - "/dev/dri:/dev/dri" - - "/dev/kfd:/dev/kfd" deploy: resources: reservations: diff --git a/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml b/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml index e9905e863..c1972f9e8 100644 --- a/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml +++ b/test/unit/data/test_compose/with_nvidia_gpu_selection.yaml @@ -11,10 +11,6 @@ services: - "8080:8080" environment: - CUDA_VISIBLE_DEVICES=0 - devices: - - "/dev/accel:/dev/accel" - - "/dev/dri:/dev/dri" - - "/dev/kfd:/dev/kfd" deploy: resources: reservations: diff --git a/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml b/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml index 8c7aed896..eae7d6d11 100644 --- a/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml +++ b/test/unit/data/test_compose/with_nvidia_gpu_vulkan_image.yaml @@ -11,10 +11,6 @@ services: - "8080:8080" environment: - CUDA_VISIBLE_DEVICES=0 - devices: - - "/dev/accel:/dev/accel" - - "/dev/dri:/dev/dri" - - "/dev/kfd:/dev/kfd" deploy: resources: reservations: diff --git a/test/unit/data/test_quadlet/basic/tinyllama.container b/test/unit/data/test_quadlet/basic/tinyllama.container index 8fab939a1..161bd3baa 100644 --- a/test/unit/data/test_quadlet/basic/tinyllama.container +++ b/test/unit/data/test_quadlet/basic/tinyllama.container @@ -3,9 +3,6 @@ Description=RamaLama tinyllama AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/draft_model/tinyllama.container b/test/unit/data/test_quadlet/draft_model/tinyllama.container index cd2c41d7f..95e8a30f5 100644 --- a/test/unit/data/test_quadlet/draft_model/tinyllama.container +++ b/test/unit/data/test_quadlet/draft_model/tinyllama.container @@ -3,9 +3,6 @@ Description=RamaLama tinyllama AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/empty/tinyllama.container b/test/unit/data/test_quadlet/empty/tinyllama.container index 8fab939a1..161bd3baa 100644 --- a/test/unit/data/test_quadlet/empty/tinyllama.container +++ b/test/unit/data/test_quadlet/empty/tinyllama.container @@ -3,9 +3,6 @@ Description=RamaLama tinyllama AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/modelfromstore/modelfromstore.container b/test/unit/data/test_quadlet/modelfromstore/modelfromstore.container index 8716fda61..8eb3f6f8f 100644 --- a/test/unit/data/test_quadlet/modelfromstore/modelfromstore.container +++ b/test/unit/data/test_quadlet/modelfromstore/modelfromstore.container @@ -3,9 +3,6 @@ Description=RamaLama modelfromstore AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/modelfromstore_add_to_unit/modelfromstore_add_to_unit.container b/test/unit/data/test_quadlet/modelfromstore_add_to_unit/modelfromstore_add_to_unit.container index a9c2b998e..837c0b9d2 100644 --- a/test/unit/data/test_quadlet/modelfromstore_add_to_unit/modelfromstore_add_to_unit.container +++ b/test/unit/data/test_quadlet/modelfromstore_add_to_unit/modelfromstore_add_to_unit.container @@ -3,9 +3,6 @@ Description=RamaLama modelfromstore_add_to_unit AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/modelfromstore_ct/modelfromstore_ct.container b/test/unit/data/test_quadlet/modelfromstore_ct/modelfromstore_ct.container index 24b7a1ffb..2b132ee9c 100644 --- a/test/unit/data/test_quadlet/modelfromstore_ct/modelfromstore_ct.container +++ b/test/unit/data/test_quadlet/modelfromstore_ct/modelfromstore_ct.container @@ -3,9 +3,6 @@ Description=RamaLama modelfromstore_ct AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/modelfromstore_mmproj/modelfromstore_mmproj.container b/test/unit/data/test_quadlet/modelfromstore_mmproj/modelfromstore_mmproj.container index ea5dbfdbe..a8b4e0aae 100644 --- a/test/unit/data/test_quadlet/modelfromstore_mmproj/modelfromstore_mmproj.container +++ b/test/unit/data/test_quadlet/modelfromstore_mmproj/modelfromstore_mmproj.container @@ -3,9 +3,6 @@ Description=RamaLama modelfromstore_mmproj AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/multipart/gpt-oss-120b.container b/test/unit/data/test_quadlet/multipart/gpt-oss-120b.container index bcb282308..76a0898cf 100644 --- a/test/unit/data/test_quadlet/multipart/gpt-oss-120b.container +++ b/test/unit/data/test_quadlet/multipart/gpt-oss-120b.container @@ -3,9 +3,6 @@ Description=RamaLama gpt-oss-120b AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/oci_basic/oci-model.container b/test/unit/data/test_quadlet/oci_basic/oci-model.container index 806548842..4485fcd63 100644 --- a/test/unit/data/test_quadlet/oci_basic/oci-model.container +++ b/test/unit/data/test_quadlet/oci_basic/oci-model.container @@ -3,9 +3,6 @@ Description=RamaLama oci-model AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/oci_port/oci-model-port.container b/test/unit/data/test_quadlet/oci_port/oci-model-port.container index e892c39b8..467280e3c 100644 --- a/test/unit/data/test_quadlet/oci_port/oci-model-port.container +++ b/test/unit/data/test_quadlet/oci_port/oci-model-port.container @@ -3,9 +3,6 @@ Description=RamaLama oci-model-port AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/oci_rag/oci-model-rag.container b/test/unit/data/test_quadlet/oci_rag/oci-model-rag.container index 3d966f567..9d6cb7fe6 100644 --- a/test/unit/data/test_quadlet/oci_rag/oci-model-rag.container +++ b/test/unit/data/test_quadlet/oci_rag/oci-model-rag.container @@ -3,9 +3,6 @@ Description=RamaLama oci-model-rag AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/data/test_quadlet/portmapping/tinyllama.container b/test/unit/data/test_quadlet/portmapping/tinyllama.container index 63b5894af..79515b4bb 100644 --- a/test/unit/data/test_quadlet/portmapping/tinyllama.container +++ b/test/unit/data/test_quadlet/portmapping/tinyllama.container @@ -3,9 +3,6 @@ Description=RamaLama tinyllama AI Model Service After=local-fs.target [Container] -AddDevice=-/dev/accel -AddDevice=-/dev/dri -AddDevice=-/dev/kfd AddDevice=nvidia.com/gpu=all Image=testimage RunInit=true diff --git a/test/unit/test_common.py b/test/unit/test_common.py index 168e9de79..4f913d0d5 100644 --- a/test/unit/test_common.py +++ b/test/unit/test_common.py @@ -26,6 +26,7 @@ ensure_image, find_in_cdi, get_accel, + get_gpu_devices, has_nvidia_vulkan_icd, host_available, host_cmd, @@ -1009,3 +1010,31 @@ def test_native_linux(self): patch("builtins.open", mock_open(read_data="7.1.13-100.fc43.x86_64\n")), ): assert not is_windows_or_wsl() + + +class TestGetGpuDevices: + devices = {"/dev/accel": "accel", "/dev/dri": "dri", "/dev/kfd": "kfd"} + + def _get_gpu_devices(self, accel_env_vars, environ=None): + with ( + patch("os.path.exists", lambda path: path in self.devices), + patch.dict("os.environ", environ or {}, clear=True), + ): + return get_gpu_devices(accel_env_vars) + + def test_host_devices(self): + assert self._get_gpu_devices({"HIP_VISIBLE_DEVICES": "0"}) == { + "accel": "/dev/accel", + "dri": "/dev/dri", + "kfd": "/dev/kfd", + } + + def test_nvidia_gets_none(self): + # The container toolkit passes the NVIDIA GPUs in itself. Anything the + # host's GPU devices would add on top is a GPU that was not asked for, + # an iGPU say, and the vulkan backend would offload onto it. + assert self._get_gpu_devices({"CUDA_VISIBLE_DEVICES": "0"}) == {} + + def test_defaults_to_the_environment(self): + assert self._get_gpu_devices(None, {"CUDA_VISIBLE_DEVICES": "0"}) == {} + assert self._get_gpu_devices(None, {"HIP_VISIBLE_DEVICES": "0"}) != {} diff --git a/test/unit/test_engine.py b/test/unit/test_engine.py index e619e3bfd..69b0b9799 100644 --- a/test/unit/test_engine.py +++ b/test/unit/test_engine.py @@ -58,6 +58,45 @@ def test_cuda_device_options_docker_selection(self): exec_args = self._cuda_device_args("docker", "1,2", selected=["1", "2"]) self.assertEqual(exec_args, ["--gpus", '"device=1,2"', "-e", "CUDA_VISIBLE_DEVICES=0,1"]) + def _host_gpu_device_args(self, accel_env_vars): + engine = ramalama.engine.Engine(self.base_args) + engine.exec_args = [] + host_devices = ["/dev/accel", "/dev/dri", "/dev/kfd"] + with ( + patch("ramalama.engine.get_accel_env_vars", return_value=dict(accel_env_vars)), + # A GPU that WSL would expose as /dev/dxg is a native one here. + patch("ramalama.engine.is_windows_or_wsl", return_value=False), + patch("os.path.exists", lambda path: path in host_devices), + patch("glob.glob", lambda path: [path] if path in host_devices else []), + patch.object(ramalama.common, "podman_machine_accel", False), + patch.object(ramalama.common, "nvidia_selected_devices", []), + ): + engine.add_device_options() + return engine.exec_args + + def test_host_gpu_devices(self): + exec_args = self._host_gpu_device_args({"HIP_VISIBLE_DEVICES": "0"}) + self.assertEqual( + exec_args, + [ + "--device", + "/dev/accel", + "--device", + "/dev/dri", + "--device", + "/dev/kfd", + "-e", + "HIP_VISIBLE_DEVICES=0", + ], + ) + + def test_host_gpu_devices_left_out_for_nvidia(self): + # The container toolkit passes the NVIDIA GPUs in itself, so /dev/dri + # could only add a GPU that was not asked for - an iGPU on a hybrid + # host - and the Vulkan backend would offload onto it. + exec_args = self._host_gpu_device_args({"CUDA_VISIBLE_DEVICES": "0"}) + self.assertEqual(exec_args, ["--device", "nvidia.com/gpu=all", "-e", "CUDA_VISIBLE_DEVICES=0"]) + def _wsl_device_args(self, accel_env_var, windows_or_wsl): engine = ramalama.engine.Engine(self.base_args) engine.exec_args = [] diff --git a/test/unit/test_quadlet.py b/test/unit/test_quadlet.py index 72a33b0a2..280b6e621 100644 --- a/test/unit/test_quadlet.py +++ b/test/unit/test_quadlet.py @@ -325,3 +325,25 @@ def test_quadlet_generate(input: Input, expected_files_path: Path, monkeypatch): del expected_files[file.filename] assert expected_files == dict() + + +def test_quadlet_nvidia_selection(monkeypatch): + """A narrowed CUDA_VISIBLE_DEVICES reserves just those GPUs, as it does for + "ramalama run": the Vulkan backend cannot be filtered any other way.""" + monkeypatch.setattr("os.path.exists", lambda path: False) + monkeypatch.setattr("ramalama.quadlet.get_accel", lambda: "cuda") + monkeypatch.setattr("ramalama.quadlet.get_accel_env_vars", lambda: {"CUDA_VISIBLE_DEVICES": "1,2"}) + monkeypatch.setattr("ramalama.common.nvidia_selected_devices", ["1", "2"]) + + files = Quadlet("tinyllama", ("/blob", "model"), None, None, Args(), [], False, None, None).generate() + with io.StringIO() as sio: + for file in files: + file._write(sio) + content = sio.getvalue() + + assert "AddDevice=nvidia.com/gpu=1" in content + assert "AddDevice=nvidia.com/gpu=2" in content + assert "AddDevice=nvidia.com/gpu=all" not in content + # The container sees the two GPUs as 0 and 1, so the host's indices would + # name a device that is not there. + assert "Environment=CUDA_VISIBLE_DEVICES=0,1" in content