@@ -81,9 +81,44 @@ func Test_SchedulingOrchestration_PodAutoscaling(t *testing.T) {
8181 }
8282 h .Logf ("Confirmed vLLM exposes vllm:num_requests_waiting metric" )
8383
84+ // Verify the custom metrics pipeline is working before generating load.
85+ h .Logf ("### Verify custom metrics pipeline" )
86+
87+ // Check that the custom metrics API is registered.
88+ h .Logf ("Checking custom metrics API registration..." )
89+ h .ShellExec ("kubectl get apiservice v1beta1.custom.metrics.k8s.io -o wide || true" )
90+
91+ // Check prometheus-adapter is running.
92+ h .Logf ("Checking prometheus-adapter deployment status..." )
93+ h .ShellExec ("kubectl get pods -n monitoring -l app.kubernetes.io/name=prometheus-adapter -o wide || true" )
94+
95+ // Query Prometheus directly to see if vllm:num_requests_waiting is being scraped.
96+ h .Logf ("Querying Prometheus for vllm:num_requests_waiting metric..." )
97+ h .ShellExec (
98+ "kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/query?query=vllm%3Anum_requests_waiting' 2>/dev/null || echo 'Failed to query Prometheus'" ,
99+ )
100+
101+ // Query Prometheus for scrape targets to see if vLLM pod is being scraped.
102+ h .Logf ("Checking Prometheus scrape targets for vLLM..." )
103+ h .ShellExec (
104+ "kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/targets?state=active' 2>/dev/null | grep -o '\" scrapePool\" :\" [^\" ]*vllm[^\" ]*\" ' || echo 'No vLLM scrape targets found'" ,
105+ )
106+
107+ // Query the custom metrics API directly to see if prometheus-adapter is serving the metric.
108+ h .Logf ("Querying custom metrics API for vllm_num_requests_waiting..." )
109+ h .ShellExec (fmt .Sprintf (
110+ "kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1/namespaces/%s/pods/*/vllm_num_requests_waiting 2>&1 || echo 'Custom metric not available via API'" ,
111+ ns ,
112+ ))
113+
114+ // Also list all available custom metrics to see what the adapter is exposing.
115+ h .Logf ("Listing all available custom metrics..." )
116+ h .ShellExec ("kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1 2>&1 || echo 'Custom metrics API not available'" )
117+
84118 // Check HPA status before load.
85119 h .Logf ("### Check initial HPA status" )
86120 h .ShellExec (fmt .Sprintf ("kubectl get hpa -n %s vllm-qwen25-500m -o wide" , ns ))
121+ h .ShellExec (fmt .Sprintf ("kubectl describe hpa -n %s vllm-qwen25-500m" , ns ))
87122
88123 // Deploy load generator.
89124 h .Logf ("## Deploy load generator" )
@@ -107,8 +142,41 @@ func Test_SchedulingOrchestration_PodAutoscaling(t *testing.T) {
107142 }
108143
109144 // Also log current metrics for debugging.
110- hpaStatus := h .ShellExec (fmt .Sprintf ("kubectl get hpa -n %s vllm-qwen25-500m -o wide" , ns ))
111- h .Logf ("Attempt %d: HPA status: %s" , attempt , hpaStatus .Stdout ())
145+ h .Logf ("Checking hpa status..." )
146+ h .ShellExec (fmt .Sprintf ("kubectl get hpa -n %s vllm-qwen25-500m -o wide" , ns ))
147+
148+ // Every 5 attempts, do deeper diagnostics on the metrics pipeline.
149+ if attempt % 5 == 1 {
150+ h .Logf ("### Diagnostics at attempt %d" , attempt )
151+
152+ // Check custom metrics API for the metric value.
153+ h .ShellExec (fmt .Sprintf (
154+ "kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1/namespaces/%s/pods/*/vllm_num_requests_waiting 2>&1 || echo 'Custom metric not available'" ,
155+ ns ,
156+ ))
157+
158+ // Query Prometheus for the raw metric.
159+ h .ShellExec (
160+ "kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/query?query=vllm%3Anum_requests_waiting' 2>/dev/null || echo 'Failed to query Prometheus'" ,
161+ )
162+
163+ // Check HPA conditions for error messages.
164+ h .ShellExec (fmt .Sprintf (
165+ "kubectl get hpa -n %s vllm-qwen25-500m -o jsonpath='{.status.conditions}' || true" ,
166+ ns ,
167+ ))
168+
169+ // Check prometheus-adapter logs for errors.
170+ h .ShellExec (
171+ "kubectl logs -n monitoring -l app.kubernetes.io/name=prometheus-adapter --tail=20 || true" ,
172+ )
173+
174+ // Check if load generator is actually running.
175+ h .ShellExec (fmt .Sprintf (
176+ "kubectl get pods -n %s -l app=vllm-load-generator -o wide || true" ,
177+ ns ,
178+ ))
179+ }
112180
113181 if attempt < maxAttempts {
114182 h .Logf ("Attempt %d: HPA has not scaled yet, waiting 30s..." , attempt )
@@ -117,8 +185,34 @@ func Test_SchedulingOrchestration_PodAutoscaling(t *testing.T) {
117185 }
118186
119187 if ! scaled {
120- // Log HPA events and conditions for debugging.
188+ // Comprehensive debugging on failure.
189+ h .Logf ("### Failure diagnostics" )
190+
121191 h .ShellExec (fmt .Sprintf ("kubectl describe hpa -n %s vllm-qwen25-500m" , ns ))
192+
193+ // Final check of custom metrics API.
194+ h .ShellExec (fmt .Sprintf (
195+ "kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1/namespaces/%s/pods/*/vllm_num_requests_waiting 2>&1 || echo 'Custom metric not available'" ,
196+ ns ,
197+ ))
198+
199+ // List all custom metrics the adapter knows about.
200+ h .ShellExec ("kubectl get --raw /apis/custom.metrics.k8s.io/v1beta1 2>&1 || echo 'Custom metrics API not available'" )
201+
202+ // Prometheus query for the raw metric.
203+ h .ShellExec (
204+ "kubectl exec -n monitoring statefulset/prometheus-kube-prometheus-stack-prometheus -c prometheus -- wget -qO- 'http://localhost:9090/api/v1/query?query=vllm%3Anum_requests_waiting' 2>/dev/null || echo 'Failed to query Prometheus'" ,
205+ )
206+
207+ // prometheus-adapter logs.
208+ h .ShellExec ("kubectl logs -n monitoring -l app.kubernetes.io/name=prometheus-adapter --tail=50 || true" )
209+
210+ // prometheus-adapter config (to verify the rule).
211+ h .ShellExec ("kubectl get configmap -n monitoring prometheus-adapter -o yaml 2>&1 || echo 'ConfigMap not found'" )
212+
213+ // Check PodMonitor was created and Prometheus discovered it.
214+ h .ShellExec (fmt .Sprintf ("kubectl get podmonitor -n %s -o yaml || true" , ns ))
215+
122216 h .Errorf ("HPA did not scale the deployment above 1 replica within the expected time" )
123217 } else {
124218 h .Success ("HPA successfully scaled vLLM deployment based on custom metric vllm_num_requests_waiting" )
0 commit comments