Sitelet https://github.com/kubernetes/kops/commit/c3f81fa60474655f986354102b3fc7672b831421
Skip to content

Commit c3f81fa

Browse files
committed
[aiconformance]: lots more debugging in pod-autoscaling test
1 parent 90d8c68 commit c3f81fa

3 files changed

Lines changed: 111 additions & 8 deletions

File tree

‎tests/e2e/scenarios/ai-conformance/validators/markdown.go‎

Lines changed: 12 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -109,10 +109,19 @@ func (o *MarkdownOutput) BeforeShellExec(command string) {
109109

110110
// AfterShellExec writes the result of the executed shell command to the markdown file in a formatted code block.
111111
func (o *MarkdownOutput) AfterShellExec(command string, results *CommandResult) {
112-
o.printf("```bash")
113-
o.printf("%s", results.Stdout())
114-
o.printf("%s", results.Stderr())
115112
o.printf("```\n")
113+
stdout := strings.TrimSpace(results.Stdout())
114+
stderr := strings.TrimSpace(results.Stderr())
115+
if stdout != "" {
116+
o.printf("%s", stdout)
117+
}
118+
if stderr != "" {
119+
if stdout != "" {
120+
o.printf("\n")
121+
}
122+
o.printf("%s", stderr)
123+
}
124+
o.printf("\n```\n")
116125

117126
if results.Err() != nil {
118127
o.printf("Error:\n```\n%v\n```\n", results.Err())

‎tests/e2e/scenarios/ai-conformance/validators/scheduling-orchestration/podautoscaling/podautoscaling_test.go‎

Lines changed: 97 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -81,9 +81,44 @@ func Test_SchedulingOrchestration_PodAutoscaling(t *testing.T) {
8181
}
8282
h.Logf("Confirmed vLLM exposes vllm:num_requests_waiting metric")
8383

84+
// Verify the custom metrics pipeline is working before generating load.
85+
h.Logf("### Verify custom metrics pipeline")
86+
87+
// Check that the custom metrics API is registered.
88+
h.Logf("Checking custom metrics API registration...")
89+
h.ShellExec("kubectl get apiservice v1beta1.custom.metrics.k8s.io -o wide || true")
90+
91+
// Check prometheus-adapter is running.
92+
h.Logf("Checking prometheus-adapter deployment status...")
93+
h.ShellExec("kubectl get pods -n monitoring -l app.kubernetes.io/name=prometheus-adapter -o wide || true")
94+
95+
// Query Prometheus directly to see if vllm:num_requests_waiting is being scraped.
96+
h.Logf("Querying Prometheus for vllm:num_requests_waiting metric...")
97+
h.ShellExec(
98+
"kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/query?query=vllm%3Anum_requests_waiting' 2>/dev/null || echo 'Failed to query Prometheus'",
99+
)
100+
101+
// Query Prometheus for scrape targets to see if vLLM pod is being scraped.
102+
h.Logf("Checking Prometheus scrape targets for vLLM...")
103+
h.ShellExec(
104+
"kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/targets?state=active' 2>/dev/null | grep -o '\"scrapePool\":\"[^\"]*vllm[^\"]*\"' || echo 'No vLLM scrape targets found'",
105+
)
106+
107+
// Query the custom metrics API directly to see if prometheus-adapter is serving the metric.
108+
h.Logf("Querying custom metrics API for vllm_num_requests_waiting...")
109+
h.ShellExec(fmt.Sprintf(
110+
"kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1/namespaces/%s/pods/*/vllm_num_requests_waiting 2>&1 || echo 'Custom metric not available via API'",
111+
ns,
112+
))
113+
114+
// Also list all available custom metrics to see what the adapter is exposing.
115+
h.Logf("Listing all available custom metrics...")
116+
h.ShellExec("kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1 2>&1 || echo 'Custom metrics API not available'")
117+
84118
// Check HPA status before load.
85119
h.Logf("### Check initial HPA status")
86120
h.ShellExec(fmt.Sprintf("kubectl get hpa -n %s vllm-qwen25-500m -o wide", ns))
121+
h.ShellExec(fmt.Sprintf("kubectl describe hpa -n %s vllm-qwen25-500m", ns))
87122

88123
// Deploy load generator.
89124
h.Logf("## Deploy load generator")
@@ -107,8 +142,41 @@ func Test_SchedulingOrchestration_PodAutoscaling(t *testing.T) {
107142
}
108143

109144
// Also log current metrics for debugging.
110-
hpaStatus := h.ShellExec(fmt.Sprintf("kubectl get hpa -n %s vllm-qwen25-500m -o wide", ns))
111-
h.Logf("Attempt %d: HPA status: %s", attempt, hpaStatus.Stdout())
145+
h.Logf("Checking hpa status...")
146+
h.ShellExec(fmt.Sprintf("kubectl get hpa -n %s vllm-qwen25-500m -o wide", ns))
147+
148+
// Every 5 attempts, do deeper diagnostics on the metrics pipeline.
149+
if attempt%5 == 1 {
150+
h.Logf("### Diagnostics at attempt %d", attempt)
151+
152+
// Check custom metrics API for the metric value.
153+
h.ShellExec(fmt.Sprintf(
154+
"kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1/namespaces/%s/pods/*/vllm_num_requests_waiting 2>&1 || echo 'Custom metric not available'",
155+
ns,
156+
))
157+
158+
// Query Prometheus for the raw metric.
159+
h.ShellExec(
160+
"kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/query?query=vllm%3Anum_requests_waiting' 2>/dev/null || echo 'Failed to query Prometheus'",
161+
)
162+
163+
// Check HPA conditions for error messages.
164+
h.ShellExec(fmt.Sprintf(
165+
"kubectl get hpa -n %s vllm-qwen25-500m -o jsonpath='{.status.conditions}' || true",
166+
ns,
167+
))
168+
169+
// Check prometheus-adapter logs for errors.
170+
h.ShellExec(
171+
"kubectl logs -n monitoring -l app.kubernetes.io/name=prometheus-adapter --tail=20 || true",
172+
)
173+
174+
// Check if load generator is actually running.
175+
h.ShellExec(fmt.Sprintf(
176+
"kubectl get pods -n %s -l app=vllm-load-generator -o wide || true",
177+
ns,
178+
))
179+
}
112180

113181
if attempt < maxAttempts {
114182
h.Logf("Attempt %d: HPA has not scaled yet, waiting 30s...", attempt)
@@ -117,8 +185,34 @@ func Test_SchedulingOrchestration_PodAutoscaling(t *testing.T) {
117185
}
118186

119187
if !scaled {
120-
// Log HPA events and conditions for debugging.
188+
// Comprehensive debugging on failure.
189+
h.Logf("### Failure diagnostics")
190+
121191
h.ShellExec(fmt.Sprintf("kubectl describe hpa -n %s vllm-qwen25-500m", ns))
192+
193+
// Final check of custom metrics API.
194+
h.ShellExec(fmt.Sprintf(
195+
"kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1/namespaces/%s/pods/*/vllm_num_requests_waiting 2>&1 || echo 'Custom metric not available'",
196+
ns,
197+
))
198+
199+
// List all custom metrics the adapter knows about.
200+
h.ShellExec("kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1 2>&1 || echo 'Custom metrics API not available'")
201+
202+
// Prometheus query for the raw metric.
203+
h.ShellExec(
204+
"kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/query?query=vllm%3Anum_requests_waiting' 2>/dev/null || echo 'Failed to query Prometheus'",
205+
)
206+
207+
// prometheus-adapter logs.
208+
h.ShellExec("kubectl logs -n monitoring -l app.kubernetes.io/name=prometheus-adapter --tail=50 || true")
209+
210+
// prometheus-adapter config (to verify the rule).
211+
h.ShellExec("kubectl get configmap -n monitoring prometheus-adapter -o yaml 2>&1 || echo 'ConfigMap not found'")
212+
213+
// Check PodMonitor was created and Prometheus discovered it.
214+
h.ShellExec(fmt.Sprintf("kubectl get podmonitor -n %s -o yaml || true", ns))
215+
122216
h.Errorf("HPA did not scale the deployment above 1 replica within the expected time")
123217
} else {
124218
h.Success("HPA successfully scaled vLLM deployment based on custom metric vllm_num_requests_waiting")

‎tests/e2e/scenarios/ai-conformance/validators/security/secure_accelerator_access/secure_accelerator_access_test.go‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -71,8 +71,8 @@ func TestSecurity_SecureAcceleratorAccess(t *testing.T) {
7171
ns := h.TestNamespace()
7272

7373
h.ApplyManifest(ns, "testdata/accelerator-isolation.yaml")
74-
h.ShellExec(fmt.Sprintf("kubectl wait -n %s --for=condition=available deployment/accelerator-isolation-1 --timeout=60s", ns))
75-
h.ShellExec(fmt.Sprintf("kubectl wait -n %s --for=condition=available deployment/accelerator-isolation-2 --timeout=60s", ns))
74+
h.ShellExec(fmt.Sprintf("kubectl wait -n %s --for=condition=available deployment/accelerator-isolation-1 --timeout=120s", ns))
75+
h.ShellExec(fmt.Sprintf("kubectl wait -n %s --for=condition=available deployment/accelerator-isolation-2 --timeout=120s", ns))
7676

7777
logs1 := h.ShellExec(fmt.Sprintf("kubectl logs -n %s deployment/accelerator-isolation-1", ns))
7878
logs2 := h.ShellExec(fmt.Sprintf("kubectl logs -n %s deployment/accelerator-isolation-2", ns))

0 commit comments

Comments
 (0)