|
| 1 | +/* |
| 2 | +Copyright The Kubernetes Authors. |
| 3 | +
|
| 4 | +Licensed under the Apache License, Version 2.0 (the "License"); |
| 5 | +you may not use this file except in compliance with the License. |
| 6 | +You may obtain a copy of the License at |
| 7 | +
|
| 8 | + http://www.apache.org/licenses/LICENSE-2.0 |
| 9 | +
|
| 10 | +Unless required by applicable law or agreed to in writing, software |
| 11 | +distributed under the License is distributed on an "AS IS" BASIS, |
| 12 | +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
| 13 | +See the License for the specific language governing permissions and |
| 14 | +limitations under the License. |
| 15 | +*/ |
| 16 | + |
| 17 | +package clusterautoscaling |
| 18 | + |
| 19 | +import ( |
| 20 | + "fmt" |
| 21 | + "strings" |
| 22 | + "testing" |
| 23 | + "time" |
| 24 | + |
| 25 | + "k8s.io/kops/tests/e2e/scenarios/ai-conformance/validators" |
| 26 | +) |
| 27 | + |
| 28 | +// Test_SchedulingOrchestration_ClusterAutoscaling verifies that the cluster autoscaler |
| 29 | +// (or equivalent mechanism) can scale up node groups containing GPU accelerators |
| 30 | +// based on pending pods requesting those accelerators. |
| 31 | +// |
| 32 | +// It counts the current number of GPU nodes, deploys N+1 replicas of a simple |
| 33 | +// GPU workload (each requesting one GPU), and verifies that the cluster |
| 34 | +// scales up to accommodate the additional pod. |
| 35 | +func Test_SchedulingOrchestration_ClusterAutoscaling(t *testing.T) { |
| 36 | + // Description: |
| 37 | + // If the platform provides a cluster autoscaler or an equivalent mechanism, |
| 38 | + // it must be able to scale up/down node groups containing specific accelerator types |
| 39 | + // based on pending pods requesting those accelerators. |
| 40 | + |
| 41 | + h := validators.NewValidatorHarness(t) |
| 42 | + |
| 43 | + h.Logf("# Cluster Autoscaling for GPU Nodes") |
| 44 | + |
| 45 | + h.Run("cluster-autoscaling-gpu", func(h *validators.ValidatorHarness) { |
| 46 | + ns := h.TestNamespace() |
| 47 | + |
| 48 | + // Count the current number of nodes with GPUs by looking at resource slices |
| 49 | + // that advertise GPU devices. |
| 50 | + h.Logf("## Determine current GPU node count") |
| 51 | + |
| 52 | + listGPUNodes := func() []string { |
| 53 | + result := h.ShellExec("kubectl get nodes -l nvidia.com/gpu.present=true -o name") |
| 54 | + nodes := strings.Split(strings.TrimSpace(result.Stdout()), "\n") |
| 55 | + return nodes |
| 56 | + } |
| 57 | + |
| 58 | + initialGPUNodes := listGPUNodes() |
| 59 | + h.Logf("Found %d GPU nodes initially (%v)", len(initialGPUNodes), initialGPUNodes) |
| 60 | + |
| 61 | + if len(initialGPUNodes) == 0 { |
| 62 | + h.Fatalf("No GPU nodes found in the cluster; cannot test cluster autoscaling for GPUs") |
| 63 | + } |
| 64 | + |
| 65 | + // Deploy the GPU probe workload with 1 replica first. |
| 66 | + h.Logf("## Deploy GPU probe workload") |
| 67 | + h.ApplyManifest(ns, "testdata/cluster-autoscaling-workload.yaml") |
| 68 | + |
| 69 | + // Scale to N+1 replicas to force the autoscaler to add a GPU node. |
| 70 | + targetReplicas := len(initialGPUNodes) + 1 |
| 71 | + h.Logf("## Scale deployment to %d replicas (initial GPU nodes: %d)", targetReplicas, len(initialGPUNodes)) |
| 72 | + h.ShellExec(fmt.Sprintf("kubectl scale deployment/cluster-autoscaling-workload -n %s --replicas=%d", ns, targetReplicas)) |
| 73 | + |
| 74 | + // Wait for at least one pod to be Pending (confirming we need a new node). |
| 75 | + h.Logf("### Verify at least one pod is pending") |
| 76 | + h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns)) |
| 77 | + |
| 78 | + // Poll for the GPU node count to increase. |
| 79 | + h.Logf("## Wait for cluster to scale up") |
| 80 | + var scaledUp bool |
| 81 | + const maxAttempts = 40 // 40 * 30s = 20 minutes |
| 82 | + for attempt := 1; attempt <= maxAttempts; attempt++ { |
| 83 | + currentGPUNodes := listGPUNodes() |
| 84 | + |
| 85 | + if len(currentGPUNodes) > len(initialGPUNodes) { |
| 86 | + h.Logf("Cluster scaled up: GPU nodes increased from %d to %d on attempt %d", len(initialGPUNodes), len(currentGPUNodes), attempt) |
| 87 | + scaledUp = true |
| 88 | + break |
| 89 | + } |
| 90 | + |
| 91 | + // Periodic diagnostics. |
| 92 | + if attempt%5 == 1 { |
| 93 | + h.Logf("### Diagnostics at attempt %d", attempt) |
| 94 | + h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns)) |
| 95 | + h.ShellExec("kubectl get nodes -o wide") |
| 96 | + } |
| 97 | + |
| 98 | + if attempt < maxAttempts { |
| 99 | + h.Logf("Attempt %d: GPU node count is still %d (need > %d), waiting 30s...", attempt, len(currentGPUNodes), len(initialGPUNodes)) |
| 100 | + time.Sleep(30 * time.Second) |
| 101 | + } |
| 102 | + } |
| 103 | + |
| 104 | + if !scaledUp { |
| 105 | + // Failure diagnostics. |
| 106 | + h.Logf("### Failure diagnostics") |
| 107 | + h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns)) |
| 108 | + h.ShellExec(fmt.Sprintf("kubectl describe pods -n %s -l app=cluster-autoscaling-workload", ns)) |
| 109 | + h.ShellExec("kubectl get nodes -o wide") |
| 110 | + h.ShellExec("kubectl describe nodes") |
| 111 | + h.Errorf("Cluster did not scale up GPU nodes within the expected time (initial: %d)", len(initialGPUNodes)) |
| 112 | + } |
| 113 | + |
| 114 | + // Verify all replicas eventually become ready. |
| 115 | + if scaledUp { |
| 116 | + h.Logf("## Wait for all replicas to be ready") |
| 117 | + result := h.ShellExec(fmt.Sprintf( |
| 118 | + "kubectl rollout status deployment/cluster-autoscaling-workload -n %s --timeout=600s", |
| 119 | + ns, |
| 120 | + )) |
| 121 | + if result.Err() != nil { |
| 122 | + h.Errorf("Deployment did not become fully ready: %v", result.Err()) |
| 123 | + } |
| 124 | + |
| 125 | + // Verify GPU pods are actually running nvidia-smi. |
| 126 | + h.Logf("### Verify GPU pods are running") |
| 127 | + h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns)) |
| 128 | + |
| 129 | + // Check logs from one of the pods to confirm GPU access. |
| 130 | + podListResult := h.ShellExec(fmt.Sprintf( |
| 131 | + "kubectl get pods -n %s -l app=cluster-autoscaling-workload -o name", |
| 132 | + ns, |
| 133 | + )) |
| 134 | + for _, podName := range strings.Split(strings.TrimSpace(podListResult.Stdout()), "\n") { |
| 135 | + h.ShellExec(fmt.Sprintf("kubectl logs -n %s %s --tail=5", ns, podName)) |
| 136 | + } |
| 137 | + |
| 138 | + h.Success("Cluster autoscaler scaled up GPU nodes from %d to accommodate %d GPU pods", len(initialGPUNodes), targetReplicas) |
| 139 | + } |
| 140 | + |
| 141 | + // Scale down and verify the cluster scales back down. |
| 142 | + h.Logf("## Scale down and verify cluster scale-down") |
| 143 | + h.ShellExec(fmt.Sprintf("kubectl scale deployment/cluster-autoscaling-workload -n %s --replicas=0", ns)) |
| 144 | + |
| 145 | + h.Logf("Waiting for cluster to scale down (this may take several minutes)...") |
| 146 | + var scaledDown bool |
| 147 | + const scaleDownMaxAttempts = 40 // 40 * 30s = 20 minutes |
| 148 | + for attempt := 1; attempt <= scaleDownMaxAttempts; attempt++ { |
| 149 | + currentGPUNodes := listGPUNodes() |
| 150 | + |
| 151 | + if len(currentGPUNodes) <= len(initialGPUNodes) { |
| 152 | + h.Logf("Cluster scaled down: GPU nodes decreased to %d on attempt %d", len(currentGPUNodes), attempt) |
| 153 | + scaledDown = true |
| 154 | + break |
| 155 | + } |
| 156 | + |
| 157 | + if attempt%5 == 1 { |
| 158 | + h.Logf("### Scale-down diagnostics at attempt %d", attempt) |
| 159 | + h.ShellExec("kubectl get nodes -o wide") |
| 160 | + } |
| 161 | + |
| 162 | + if attempt < scaleDownMaxAttempts { |
| 163 | + h.Logf("Attempt %d: GPU node count is still %d (need <= %d), waiting 30s...", attempt, len(currentGPUNodes), len(initialGPUNodes)) |
| 164 | + time.Sleep(30 * time.Second) |
| 165 | + } |
| 166 | + } |
| 167 | + |
| 168 | + if !scaledDown { |
| 169 | + h.Logf("### Scale-down failure diagnostics") |
| 170 | + h.ShellExec("kubectl get nodes -o wide") |
| 171 | + h.ShellExec("kubectl describe nodes") |
| 172 | + h.Errorf("Cluster did not scale down GPU nodes within the expected time") |
| 173 | + } else { |
| 174 | + h.Success("Cluster autoscaler scaled down GPU nodes back to %d", len(initialGPUNodes)) |
| 175 | + } |
| 176 | + }) |
| 177 | + |
| 178 | + if h.AllPassed() { |
| 179 | + h.RecordConformance("schedulingOrchestration", "cluster_autoscaling") |
| 180 | + } |
| 181 | +} |
0 commit comments