Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 15 additions & 1 deletion api/nvidia/v1alpha1/gpucluster_types.go
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,7 @@ type GPUClusterSpec struct {
DCGMExporter *nvidiav1.DCGMExporterSpec `json:"dcgmExporter,omitempty"`

// HostPaths defines the host paths used in host-path volumes for various components.
HostPaths nvidiav1.HostPathsSpec `json:"hostPaths,omitempty"`
HostPaths HostPathsSpec `json:"hostPaths,omitempty"`

// Daemonsets defines the common configuration applied to all DaemonSets deployed
// by the GPUCluster controller.
Expand Down Expand Up @@ -154,6 +154,20 @@ type DRADriverControllerSpec struct {
Resources *nvidiav1.ResourceRequirements `json:"resources,omitempty"`
}

// HostPathsSpec defines various paths on the host needed by GPU Operator components.
// Unlike the v1 ClusterPolicy struct it mirrors, it has no RootFS: the host root is
// hard-coded to "/" for the DRA stack.
type HostPathsSpec struct {
// DriverInstallDir represents the root at which driver files including libraries,
// config files, and executables can be found.
DriverInstallDir string `json:"driverInstallDir,omitempty"`

// KubeletRootDir represents the location of the kubelet root directory.
// If empty, it will default to "/var/lib/kubelet".
// +kubebuilder:default="/var/lib/kubelet"
KubeletRootDir string `json:"kubeletRootDir,omitempty"`
}

// GPUClusterStatus defines the observed state of GPUCluster
type GPUClusterStatus struct {
// +kubebuilder:validation:Enum=ignored;ready;notReady;disabled
Expand Down
15 changes: 15 additions & 0 deletions api/nvidia/v1alpha1/zz_generated.deepcopy.go

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

8 changes: 0 additions & 8 deletions bundle/manifests/nvidia.com_gpuclusters.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -955,14 +955,6 @@ spec:
KubeletRootDir represents the location of the kubelet root directory.
If empty, it will default to "/var/lib/kubelet".
type: string
rootFS:
description: |-
RootFS represents the path to the root filesystem of the host.
This is used by components that need to interact with the host filesystem
and as such this must be a chroot-able filesystem.
Examples include the MIG Manager and Toolkit Container which may need to
stop, start, or restart systemd services.
type: string
type: object
required:
- draDriver
Expand Down
11 changes: 6 additions & 5 deletions cmd/gpu-operator/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -193,11 +193,12 @@ func main() {
WithRestartOnlyPredicate(predicates.DriverPodRestartOnly(upgradeLogger))

if err = (&controllers.UpgradeReconciler{
Client: mgr.GetClient(),
Log: upgradeLogger,
Scheme: mgr.GetScheme(),
StateManager: clusterUpgradeStateManager,
OperatorMetrics: operatorMetrics,
Client: mgr.GetClient(),
Log: upgradeLogger,
Scheme: mgr.GetScheme(),
StateManager: clusterUpgradeStateManager,
OperatorMetrics: operatorMetrics,
OperatorNamespace: operatorNamespace,
}).SetupWithManager(ctx, mgr); err != nil {
setupLog.Error(err, "unable to create controller", "controller", "Upgrade")
os.Exit(1)
Expand Down
8 changes: 0 additions & 8 deletions config/crd/bases/nvidia.com_gpuclusters.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -955,14 +955,6 @@ spec:
KubeletRootDir represents the location of the kubelet root directory.
If empty, it will default to "/var/lib/kubelet".
type: string
rootFS:
description: |-
RootFS represents the path to the root filesystem of the host.
This is used by components that need to interact with the host filesystem
and as such this must be a chroot-able filesystem.
Examples include the MIG Manager and Toolkit Container which may need to
stop, start, or restart systemd services.
type: string
type: object
required:
- draDriver
Expand Down
33 changes: 33 additions & 0 deletions controllers/gpucluster_controller.go
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@ import (
"sigs.k8s.io/controller-runtime/pkg/reconcile"
"sigs.k8s.io/controller-runtime/pkg/source"

gpuv1 "github.com/NVIDIA/gpu-operator/api/nvidia/v1"
nvidiav1alpha1 "github.com/NVIDIA/gpu-operator/api/nvidia/v1alpha1"
"github.com/NVIDIA/gpu-operator/controllers/clusterinfo"
"github.com/NVIDIA/gpu-operator/internal/conditions"
Expand Down Expand Up @@ -120,6 +121,21 @@ func (r *GPUClusterReconciler) Reconcile(ctx context.Context, req ctrl.Request)
}
r.singleton = instance

// DRA requires all driver management through NVIDIADriver CRs: surface an unmet
// prerequisite on this CR's status and hold off deploying operands until it is met.
if msg, err := r.validatePrerequisites(ctx); err != nil {
return ctrl.Result{}, err
} else if msg != "" {
logger.V(consts.LogLevelWarning).Info("GPUCluster prerequisite not met", "reason", msg)
if err := r.updateCRStatus(ctx, instance, nvidiav1alpha1.NotReady); err != nil {
return ctrl.Result{}, err
}
if condErr := r.conditionUpdater.SetConditionsError(ctx, instance, conditions.PrerequisiteNotMet, msg); condErr != nil {
logger.Error(condErr, "failed to set condition")
}
return ctrl.Result{RequeueAfter: time.Minute}, nil
}

// The operand states render ResourceClaimTemplates with adminAccess: true, which the
// kube-scheduler only admits from a labeled namespace; label it before syncing states.
if err := r.ensureAdminAccessLabel(ctx); err != nil {
Expand Down Expand Up @@ -162,6 +178,23 @@ func (r *GPUClusterReconciler) Reconcile(ctx context.Context, req ctrl.Request)
return ctrl.Result{RequeueAfter: time.Minute}, nil
}

// validatePrerequisites checks the cross-CR rules that gate DRA enablement, returning
// a message describing the first unmet prerequisite or an empty string when all are met.
func (r *GPUClusterReconciler) validatePrerequisites(ctx context.Context) (string, error) {
clusterPolicies := &gpuv1.ClusterPolicyList{}
if err := r.List(ctx, clusterPolicies); err != nil {
return "", fmt.Errorf("error listing ClusterPolicy objects: %w", err)
}
// TODO: check only the active singleton ClusterPolicy once the singleton
// selection is resolvable across controllers (see resolveActiveConfig).
for _, clusterPolicy := range clusterPolicies.Items {
if !clusterPolicy.Spec.Driver.UseNvidiaDriverCRDType() {
return fmt.Sprintf("ClusterPolicy %s does not have driver.useNvidiaDriverCRD enabled; migrate driver management to NVIDIADriver CRs before enabling DRA", clusterPolicy.Name), nil
}
}
return "", nil
}

// ensureAdminAccessLabel patches the operator namespace with the label required by the
// kube-scheduler to allow adminAccess: true in ResourceClaim/ResourceClaimTemplate
// objects. The label is deliberately never removed: it is namespace-level configuration
Expand Down
56 changes: 49 additions & 7 deletions controllers/gpucluster_controller_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ import (
"context"
"sort"
"testing"
"time"

"github.com/stretchr/testify/require"
appsv1 "k8s.io/api/apps/v1"
Expand Down Expand Up @@ -71,14 +72,17 @@ func newGPUClusterReconciler(t *testing.T, objs ...client.Object) (*GPUClusterRe
}

// fakeStateManager returns canned SyncState results so the controller tests don't load
// real manifests. GetWatchSources is promoted from the embedded (nil) interface and is
// never called here — only SetupWithManager calls it, which these tests skip.
// real manifests. It records the last info catalog passed to SyncState so tests can
// assert on its entries. GetWatchSources is promoted from the embedded (nil) interface
// and is never called here — only SetupWithManager calls it, which these tests skip.
type fakeStateManager struct {
state.Manager
results state.Results
results state.Results
lastCatalog state.InfoCatalog
}

func (f *fakeStateManager) SyncState(_ context.Context, _ interface{}, _ state.InfoCatalog) state.Results {
func (f *fakeStateManager) SyncState(_ context.Context, _ interface{}, catalog state.InfoCatalog) state.Results {
f.lastCatalog = catalog
return f.results
}

Expand Down Expand Up @@ -213,18 +217,56 @@ func TestGPUClusterTeardownDrainsClaimConsumersFirst(t *testing.T) {
require.NoError(t, c.Get(t.Context(), types.NamespacedName{Name: plugin.Name, Namespace: "test-namespace"}, ds))
}

// A ClusterPolicy in the cluster does not disable the GPUCluster: the two stacks
// coexist, with per-node ownership decided by the nvidia.com/gpu-operator.resource-allocation.mode label.
// A ClusterPolicy in the cluster does not disable the GPUCluster, provided it
// delegates driver management to NVIDIADriver CRs: the two stacks coexist, with
// per-node ownership decided by the nvidia.com/gpu-operator.resource-allocation.mode label.
func TestGPUClusterCoexistsWithClusterPolicy(t *testing.T) {
cfg := &nvidiav1alpha1.GPUCluster{ObjectMeta: metav1.ObjectMeta{Name: "config"}}
cp := &gpuv1.ClusterPolicy{ObjectMeta: metav1.ObjectMeta{Name: "cluster-policy"}}
cp := &gpuv1.ClusterPolicy{
ObjectMeta: metav1.ObjectMeta{Name: "cluster-policy"},
Spec: gpuv1.ClusterPolicySpec{
Driver: gpuv1.DriverSpec{UseNvidiaDriverCRD: ptr.To(true)},
},
}
r, c := newGPUClusterReconciler(t, cfg, cp)

gccReconcile(t, r, cfg.Name)

require.Equal(t, nvidiav1alpha1.Ready, gccState(t, c, cfg.Name))
}

// A ClusterPolicy that manages its own driver (useNvidiaDriverCRD=false) is an invalid
// companion for DRA: the GPUCluster reports the unmet prerequisite and deploys nothing.
func TestGPUClusterClusterPolicyDriverPrerequisite(t *testing.T) {
cfg := &nvidiav1alpha1.GPUCluster{ObjectMeta: metav1.ObjectMeta{Name: "config"}}
cp := &gpuv1.ClusterPolicy{ObjectMeta: metav1.ObjectMeta{Name: "cluster-policy"}}
r, c := newGPUClusterReconciler(t, cfg, cp)
r.conditionUpdater = conditions.NewGPUClusterUpdater(c)

res, err := r.Reconcile(t.Context(), gccRequest(cfg.Name))
require.NoError(t, err)
require.Equal(t, time.Minute, res.RequeueAfter)

require.Equal(t, nvidiav1alpha1.NotReady, gccState(t, c, cfg.Name))
require.Nil(t, r.stateManager.(*fakeStateManager).lastCatalog, "operands must not be synced")

instance := &nvidiav1alpha1.GPUCluster{}
require.NoError(t, c.Get(t.Context(), types.NamespacedName{Name: cfg.Name}, instance))
cond := meta.FindStatusCondition(instance.Status.Conditions, conditions.Error)
require.NotNil(t, cond)
require.Equal(t, conditions.PrerequisiteNotMet, cond.Reason)
require.Contains(t, cond.Message, "useNvidiaDriverCRD")

// Toggling the flag to true clears the prerequisite on the next reconcile.
updated := &gpuv1.ClusterPolicy{}
require.NoError(t, c.Get(t.Context(), types.NamespacedName{Name: cp.Name}, updated))
updated.Spec.Driver.UseNvidiaDriverCRD = ptr.To(true)
require.NoError(t, c.Update(t.Context(), updated))

gccReconcile(t, r, cfg.Name)
require.Equal(t, nvidiav1alpha1.Ready, gccState(t, c, cfg.Name))
}

// First-reconciled wins (mirroring ClusterPolicy): whichever instance reconciles first
// claims ownership, regardless of name or creationTimestamp.
func TestGPUClusterSingleton(t *testing.T) {
Expand Down
69 changes: 29 additions & 40 deletions controllers/nvidiadriver_controller.go
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,13 @@ import (
"github.com/NVIDIA/gpu-operator/internal/validator"
)

// defaultHostRoot is the host root mounted into the driver daemonset. It is
// hard-coded to "/" to avoid a cross-CR dependency; only the legacy
// ClusterPolicy-only path reads spec.hostPaths.rootFS instead.
// TODO: support a custom host root via a new NVIDIADriver API field or an
// operator environment variable.
const defaultHostRoot = "/"

// NVIDIADriverReconciler reconciles a NVIDIADriver object
type NVIDIADriverReconciler struct {
client.Client
Expand All @@ -64,6 +71,7 @@ type NVIDIADriverReconciler struct {
//+kubebuilder:rbac:groups=nvidia.com,resources=nvidiadrivers,verbs=get;list;watch;create;update;patch;delete
//+kubebuilder:rbac:groups=nvidia.com,resources=nvidiadrivers/status,verbs=get;update;patch
//+kubebuilder:rbac:groups=nvidia.com,resources=nvidiadrivers/finalizers,verbs=update
//+kubebuilder:rbac:groups=nvidia.com,resources=gpuclusters,verbs=get;list;watch

// Reconcile is part of the main kubernetes reconciliation loop which aims to
// move the current state of the cluster closer to the desired state.
Expand Down Expand Up @@ -98,53 +106,33 @@ func (r *NVIDIADriverReconciler) Reconcile(ctx context.Context, req ctrl.Request
return reconcile.Result{}, nil
}

// Get the singleton NVIDIA ClusterPolicy object in the cluster.
clusterPolicyList := &gpuv1.ClusterPolicyList{}
if err := r.List(ctx, clusterPolicyList); err != nil {
wrappedErr := fmt.Errorf("error getting ClusterPolicy list: %w", err)
logger.Error(err, "error getting ClusterPolicy list")
// Resolve the active cluster configuration (ClusterPolicy, GPUCluster, or both).
clusterPolicy, gpuCluster, err := resolveActiveConfig(ctx, r.Client)
if err != nil {
logger.Error(err, "error resolving active cluster configuration")
instance.Status.State = nvidiav1alpha1.NotReady
if condErr := r.conditionUpdater.SetConditionsError(ctx, instance, conditions.ReconcileFailed, err.Error()); condErr != nil {
logger.Error(condErr, "failed to set condition")
}
return reconcile.Result{}, wrappedErr
return reconcile.Result{}, err
}

if len(clusterPolicyList.Items) == 0 {
err := fmt.Errorf("no ClusterPolicy object found in the cluster")
logger.Error(err, "failed to get ClusterPolicy object")
instance.Status.State = nvidiav1alpha1.NotReady
if condErr := r.conditionUpdater.SetConditionsError(ctx, instance, conditions.ReconcileFailed, err.Error()); condErr != nil {
// A ClusterPolicy that does not delegate its driver via useNvidiaDriverCRD
// disables reconciliation even when a GPUCluster exists, since DRA requires
// driver management through NVIDIADriver CRs.
if clusterPolicy != nil && !clusterPolicy.Spec.Driver.UseNvidiaDriverCRDType() {
msg := "useNvidiaDriverCRD is not enabled in ClusterPolicy"
logger.V(consts.LogLevelWarning).Info("NVIDIADriver reconciliation skipped", "reason", msg)
instance.Status.State = nvidiav1alpha1.Disabled
if condErr := r.conditionUpdater.SetConditionsError(ctx, instance, conditions.Reconciled, msg); condErr != nil {
logger.Error(condErr, "failed to set condition")
}
return reconcile.Result{}, err
return reconcile.Result{}, nil
}
clusterPolicyInstance := clusterPolicyList.Items[0]

// Ensure the NVIDIADriver CR has a consumer: either the ClusterPolicy delegates its
// driver to the NVIDIADriver CRD, or a GPUCluster exists. GPUCluster does
// not manage the driver itself — it is either preinstalled on the host (no NVIDIADriver
// CR) or installed via NVIDIADriver CRs, so any CR that exists alongside one is in use.
if !clusterPolicyInstance.Spec.Driver.UseNvidiaDriverCRDType() {
gpuClusters := &nvidiav1alpha1.GPUClusterList{}
if err := r.List(ctx, gpuClusters); err != nil {
wrappedErr := fmt.Errorf("error getting GPUCluster list: %w", err)
logger.Error(err, "error getting GPUCluster list")
instance.Status.State = nvidiav1alpha1.NotReady
if condErr := r.conditionUpdater.SetConditionsError(ctx, instance, conditions.ReconcileFailed, err.Error()); condErr != nil {
logger.Error(condErr, "failed to set condition")
}
return reconcile.Result{}, wrappedErr
}
if len(gpuClusters.Items) == 0 {
msg := "useNvidiaDriverCRD is not enabled in ClusterPolicy and no GPUCluster exists"
logger.V(consts.LogLevelWarning).Info("NVIDIADriver reconciliation skipped", "reason", msg)
instance.Status.State = nvidiav1alpha1.Disabled
if condErr := r.conditionUpdater.SetConditionsError(ctx, instance, conditions.Reconciled, msg); condErr != nil {
logger.Error(condErr, "failed to set condition")
}
return reconcile.Result{}, nil
}

hostRoot := defaultHostRoot
if clusterPolicy != nil && gpuCluster == nil {
hostRoot = clusterPolicy.Spec.HostPaths.RootFS
}

// Create a new InfoCatalog which is a generic interface for passing information to state managers
Expand All @@ -153,8 +141,8 @@ func (r *NVIDIADriverReconciler) Reconcile(ctx context.Context, req ctrl.Request
// Add an entry for ClusterInfo, which was collected before the NVIDIADriver controller was started
infoCatalog.Add(state.InfoTypeClusterInfo, r.ClusterInfo)

// Add an entry for Clusterpolicy, which is needed to deploy the driver daemonset
infoCatalog.Add(state.InfoTypeClusterPolicyCR, clusterPolicyInstance)
// Add the host root, which is needed to deploy the driver daemonset
infoCatalog.Add(state.InfoTypeHostRoot, hostRoot)

// Verify the nodeSelector configured for this NVIDIADriver instance does
// not conflict with any other instances. This ensures only one driver
Expand Down Expand Up @@ -405,6 +393,7 @@ func (r *NVIDIADriverReconciler) SetupWithManager(ctx context.Context, mgr ctrl.
gpuClusterMapFn := func(ctx context.Context, _ *nvidiav1alpha1.GPUCluster) []reconcile.Request {
return r.enqueueAllNVIDIADrivers(ctx)
}

err = c.Watch(
source.Kind(
mgr.GetCache(),
Expand Down
Loading
Loading