Sitelet https://github.com/kubernetes/kops/commit/29aa401a6fe011d92e646671ab557cc43844a07e
Skip to content

Commit 29aa401

Browse files
committed
[aiconformance] add test for schedulingOrchestration clusterAutoscaling
1 parent d25b7f8 commit 29aa401

3 files changed

Lines changed: 229 additions & 4 deletions

File tree

‎tests/e2e/scenarios/ai-conformance/validators/output.go‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -60,16 +60,16 @@ func (h *ValidatorHarness) Log(s string) {
6060
func (h *ValidatorHarness) Fatalf(format string, args ...interface{}) {
6161
s := fmt.Sprintf(format, args...)
6262

63-
h.output.WriteText("FATAL: " + s)
64-
h.t.Fatalf(format, args...)
63+
h.output.WriteText("FAIL: " + s)
64+
h.t.Fatalf("FAIL: "+format, args...)
6565
}
6666

6767
// Errorf is like t.Errorf, but also writes to the sinks.
6868
func (h *ValidatorHarness) Errorf(format string, args ...interface{}) {
6969
s := fmt.Sprintf(format, args...)
7070

71-
h.output.WriteText("ERROR: " + s)
72-
h.t.Errorf(format, args...)
71+
h.output.WriteText("FAIL: " + s)
72+
h.t.Errorf("FAIL: "+format, args...)
7373
}
7474

7575
// Run is like t.Run, but creates a sub-harness that shares the output.
Lines changed: 181 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,181 @@
1+
/*
2+
Copyright The Kubernetes Authors.
3+
4+
Licensed under the Apache License, Version 2.0 (the "License");
5+
you may not use this file except in compliance with the License.
6+
You may obtain a copy of the License at
7+
8+
http://www.apache.org/licenses/LICENSE-2.0
9+
10+
Unless required by applicable law or agreed to in writing, software
11+
distributed under the License is distributed on an "AS IS" BASIS,
12+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13+
See the License for the specific language governing permissions and
14+
limitations under the License.
15+
*/
16+
17+
package clusterautoscaling
18+
19+
import (
20+
"fmt"
21+
"strings"
22+
"testing"
23+
"time"
24+
25+
"k8s.io/kops/tests/e2e/scenarios/ai-conformance/validators"
26+
)
27+
28+
// Test_SchedulingOrchestration_ClusterAutoscaling verifies that the cluster autoscaler
29+
// (or equivalent mechanism) can scale up node groups containing GPU accelerators
30+
// based on pending pods requesting those accelerators.
31+
//
32+
// It counts the current number of GPU nodes, deploys N+1 replicas of a simple
33+
// GPU workload (each requesting one GPU), and verifies that the cluster
34+
// scales up to accommodate the additional pod.
35+
func Test_SchedulingOrchestration_ClusterAutoscaling(t *testing.T) {
36+
// Description:
37+
// If the platform provides a cluster autoscaler or an equivalent mechanism,
38+
// it must be able to scale up/down node groups containing specific accelerator types
39+
// based on pending pods requesting those accelerators.
40+
41+
h := validators.NewValidatorHarness(t)
42+
43+
h.Logf("# Cluster Autoscaling for GPU Nodes")
44+
45+
h.Run("cluster-autoscaling-gpu", func(h *validators.ValidatorHarness) {
46+
ns := h.TestNamespace()
47+
48+
// Count the current number of nodes with GPUs by looking at resource slices
49+
// that advertise GPU devices.
50+
h.Logf("## Determine current GPU node count")
51+
52+
listGPUNodes := func() []string {
53+
result := h.ShellExec("kubectl get nodes -l nvidia.com/gpu.present=true -o name")
54+
nodes := strings.Split(strings.TrimSpace(result.Stdout()), "\n")
55+
return nodes
56+
}
57+
58+
initialGPUNodes := listGPUNodes()
59+
h.Logf("Found %d GPU nodes initially (%v)", len(initialGPUNodes), initialGPUNodes)
60+
61+
if len(initialGPUNodes) == 0 {
62+
h.Fatalf("No GPU nodes found in the cluster; cannot test cluster autoscaling for GPUs")
63+
}
64+
65+
// Deploy the GPU probe workload with 1 replica first.
66+
h.Logf("## Deploy GPU probe workload")
67+
h.ApplyManifest(ns, "testdata/cluster-autoscaling-workload.yaml")
68+
69+
// Scale to N+1 replicas to force the autoscaler to add a GPU node.
70+
targetReplicas := len(initialGPUNodes) + 1
71+
h.Logf("## Scale deployment to %d replicas (initial GPU nodes: %d)", targetReplicas, len(initialGPUNodes))
72+
h.ShellExec(fmt.Sprintf("kubectl scale deployment/cluster-autoscaling-workload -n %s --replicas=%d", ns, targetReplicas))
73+
74+
// Wait for at least one pod to be Pending (confirming we need a new node).
75+
h.Logf("### Verify at least one pod is pending")
76+
h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns))
77+
78+
// Poll for the GPU node count to increase.
79+
h.Logf("## Wait for cluster to scale up")
80+
var scaledUp bool
81+
const maxAttempts = 40 // 40 * 30s = 20 minutes
82+
for attempt := 1; attempt <= maxAttempts; attempt++ {
83+
currentGPUNodes := listGPUNodes()
84+
85+
if len(currentGPUNodes) > len(initialGPUNodes) {
86+
h.Logf("Cluster scaled up: GPU nodes increased from %d to %d on attempt %d", len(initialGPUNodes), len(currentGPUNodes), attempt)
87+
scaledUp = true
88+
break
89+
}
90+
91+
// Periodic diagnostics.
92+
if attempt%5 == 1 {
93+
h.Logf("### Diagnostics at attempt %d", attempt)
94+
h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns))
95+
h.ShellExec("kubectl get nodes -o wide")
96+
}
97+
98+
if attempt < maxAttempts {
99+
h.Logf("Attempt %d: GPU node count is still %d (need > %d), waiting 30s...", attempt, len(currentGPUNodes), len(initialGPUNodes))
100+
time.Sleep(30 * time.Second)
101+
}
102+
}
103+
104+
if !scaledUp {
105+
// Failure diagnostics.
106+
h.Logf("### Failure diagnostics")
107+
h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns))
108+
h.ShellExec(fmt.Sprintf("kubectl describe pods -n %s -l app=cluster-autoscaling-workload", ns))
109+
h.ShellExec("kubectl get nodes -o wide")
110+
h.ShellExec("kubectl describe nodes")
111+
h.Errorf("Cluster did not scale up GPU nodes within the expected time (initial: %d)", len(initialGPUNodes))
112+
}
113+
114+
// Verify all replicas eventually become ready.
115+
if scaledUp {
116+
h.Logf("## Wait for all replicas to be ready")
117+
result := h.ShellExec(fmt.Sprintf(
118+
"kubectl rollout status deployment/cluster-autoscaling-workload -n %s --timeout=600s",
119+
ns,
120+
))
121+
if result.Err() != nil {
122+
h.Errorf("Deployment did not become fully ready: %v", result.Err())
123+
}
124+
125+
// Verify GPU pods are actually running nvidia-smi.
126+
h.Logf("### Verify GPU pods are running")
127+
h.ShellExec(fmt.Sprintf("kubectl get pods -n %s -l app=cluster-autoscaling-workload -o wide", ns))
128+
129+
// Check logs from one of the pods to confirm GPU access.
130+
podListResult := h.ShellExec(fmt.Sprintf(
131+
"kubectl get pods -n %s -l app=cluster-autoscaling-workload -o name",
132+
ns,
133+
))
134+
for _, podName := range strings.Split(strings.TrimSpace(podListResult.Stdout()), "\n") {
135+
h.ShellExec(fmt.Sprintf("kubectl logs -n %s %s --tail=5", ns, podName))
136+
}
137+
138+
h.Success("Cluster autoscaler scaled up GPU nodes from %d to accommodate %d GPU pods", len(initialGPUNodes), targetReplicas)
139+
}
140+
141+
// Scale down and verify the cluster scales back down.
142+
h.Logf("## Scale down and verify cluster scale-down")
143+
h.ShellExec(fmt.Sprintf("kubectl scale deployment/cluster-autoscaling-workload -n %s --replicas=0", ns))
144+
145+
h.Logf("Waiting for cluster to scale down (this may take several minutes)...")
146+
var scaledDown bool
147+
const scaleDownMaxAttempts = 40 // 40 * 30s = 20 minutes
148+
for attempt := 1; attempt <= scaleDownMaxAttempts; attempt++ {
149+
currentGPUNodes := listGPUNodes()
150+
151+
if len(currentGPUNodes) <= len(initialGPUNodes) {
152+
h.Logf("Cluster scaled down: GPU nodes decreased to %d on attempt %d", len(currentGPUNodes), attempt)
153+
scaledDown = true
154+
break
155+
}
156+
157+
if attempt%5 == 1 {
158+
h.Logf("### Scale-down diagnostics at attempt %d", attempt)
159+
h.ShellExec("kubectl get nodes -o wide")
160+
}
161+
162+
if attempt < scaleDownMaxAttempts {
163+
h.Logf("Attempt %d: GPU node count is still %d (need <= %d), waiting 30s...", attempt, len(currentGPUNodes), len(initialGPUNodes))
164+
time.Sleep(30 * time.Second)
165+
}
166+
}
167+
168+
if !scaledDown {
169+
h.Logf("### Scale-down failure diagnostics")
170+
h.ShellExec("kubectl get nodes -o wide")
171+
h.ShellExec("kubectl describe nodes")
172+
h.Errorf("Cluster did not scale down GPU nodes within the expected time")
173+
} else {
174+
h.Success("Cluster autoscaler scaled down GPU nodes back to %d", len(initialGPUNodes))
175+
}
176+
})
177+
178+
if h.AllPassed() {
179+
h.RecordConformance("schedulingOrchestration", "cluster_autoscaling")
180+
}
181+
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,44 @@
1+
# Deployment that requests a GPU via DRA and runs nvidia-smi periodically.
2+
# The test creates this with N+1 replicas (where N is the current GPU node count),
3+
# forcing the cluster autoscaler to provision an additional GPU node.
4+
5+
6+
apiVersion: apps/v1
7+
kind: Deployment
8+
metadata:
9+
name: cluster-autoscaling-workload
10+
labels:
11+
app: cluster-autoscaling-workload
12+
spec:
13+
# replicas is set dynamically by the test via kubectl scale
14+
replicas: 1
15+
selector:
16+
matchLabels:
17+
app: cluster-autoscaling-workload
18+
template:
19+
metadata:
20+
labels:
21+
app: cluster-autoscaling-workload
22+
spec:
23+
terminationGracePeriodSeconds: 5
24+
tolerations:
25+
- key: "nvidia.com/gpu"
26+
operator: "Exists"
27+
effect: "NoSchedule"
28+
containers:
29+
- name: cluster-autoscaling-workload
30+
image: nvcr.io/nvidia/k8s/cuda-sample:vectoradd-cuda12.5.0
31+
command:
32+
- "/bin/sh"
33+
- "-c"
34+
- |
35+
while true; do
36+
echo "$(date -Iseconds) GPU probe alive on $(hostname)"
37+
nvidia-smi --query-gpu=name,temperature.gpu,utilization.gpu,memory.used,memory.total --format=csv,noheader
38+
sleep 60
39+
done
40+
resources:
41+
requests:
42+
nvidia.com/gpu: 1
43+
limits:
44+
nvidia.com/gpu: 1

0 commit comments

Comments
 (0)