-
Notifications
You must be signed in to change notification settings - Fork 4.7k
[aiconformance]: Add accelerator metrics validator test #18064
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Merged
k8s-ci-robot
merged 1 commit into
kubernetes:master
from
justinsb:aiconformance_accelerator_metrics
Mar 14, 2026
Merged
Changes from all commits
Commits
File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
There are no files selected for viewing
131 changes: 131 additions & 0 deletions
131
...s/ai-conformance/validators/observability/accelerator_metrics/accelerator_metrics_test.go
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,131 @@ | ||
| /* | ||
| Copyright The Kubernetes Authors. | ||
|
|
||
| Licensed under the Apache License, Version 2.0 (the "License"); | ||
| you may not use this file except in compliance with the License. | ||
| You may obtain a copy of the License at | ||
|
|
||
| http://www.apache.org/licenses/LICENSE-2.0 | ||
|
|
||
| Unless required by applicable law or agreed to in writing, software | ||
| distributed under the License is distributed on an "AS IS" BASIS, | ||
| WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| See the License for the specific language governing permissions and | ||
| limitations under the License. | ||
| */ | ||
|
|
||
| package ai_inference | ||
|
|
||
| import ( | ||
| "fmt" | ||
| "strings" | ||
| "testing" | ||
| "time" | ||
|
|
||
| "k8s.io/kops/tests/e2e/scenarios/ai-conformance/validators" | ||
| ) | ||
|
|
||
| // TestObservability_AcceleratorMetrics corresponds to the observability/accelerator_metrics conformance requirement. | ||
| func TestObservability_AcceleratorMetrics(t *testing.T) { | ||
| // Description: | ||
| // For supported accelerator types, the platform must allow for the installation and successful operation of at least one accelerator metrics solution | ||
| // that exposes fine-grained performance metrics via a standardized, machine-readable metrics endpoint. | ||
| // This must include a core set of metrics for per-accelerator utilization and memory usage. | ||
| // Additionally, other relevant metrics such as temperature, power draw, and interconnect bandwidth should be exposed | ||
| // if the underlying hardware or virtualization layer makes them available. | ||
| // The list of metrics should align with emerging standards, such as OpenTelemetry metrics, to ensure interoperability. | ||
| // The platform may provide a managed solution, but this is not required for conformance." | ||
|
|
||
| h := validators.NewValidatorHarness(t) | ||
|
|
||
| h.Logf("# Observability: Accelerator Metrics") | ||
|
|
||
| h.Run("nvidia-metrics", func(h *validators.ValidatorHarness) { | ||
| h.Logf("## Verify NVIDIA Metrics") | ||
|
|
||
| h.ShellExec("kubectl get service -n gpu-operator") | ||
|
|
||
| ns := h.TestNamespace() | ||
|
|
||
| requiredMetrics := []string{ | ||
| "DCGM_FI_DEV_GPU_TEMP", | ||
| "DCGM_FI_DEV_POWER_USAGE", | ||
| "DCGM_FI_DEV_GPU_UTIL", | ||
| "DCGM_FI_DEV_FB_USED", | ||
| } | ||
|
|
||
| var metricClasses map[string]bool | ||
| var metricsOutput string | ||
|
|
||
| // Retry scraping metrics, as the DCGM exporter may not have completed its first collection cycle yet. | ||
| const maxAttempts = 5 | ||
| for attempt := 1; attempt <= maxAttempts; attempt++ { | ||
| podName := fmt.Sprintf("scrape-accelerator-metrics-%d", attempt) | ||
|
|
||
| h.ShellExec(fmt.Sprintf( | ||
| "kubectl run %s -n %s --image=registry.k8s.io/e2e-test-images/agnhost:2.39 --restart=Never --command -- curl -sS http://nvidia-dcgm-exporter.gpu-operator.svc.cluster.local:9400/metrics", | ||
| podName, ns, | ||
| )) | ||
| //h.ShellExec(fmt.Sprintf("kubectl wait -n %s pod/%s --for=condition=Ready --timeout=60s", ns, jobName)) | ||
| h.ShellExec(fmt.Sprintf("kubectl wait -n %s pod/%s --for=jsonpath='{.status.phase}'=Succeeded --timeout=120s", ns, podName)) | ||
|
|
||
| logs := h.ShellExec(fmt.Sprintf("kubectl logs -n %s %s", ns, podName)) | ||
| metricsOutput = logs.Stdout() | ||
|
|
||
| metricClasses = make(map[string]bool) | ||
| for _, line := range strings.Split(metricsOutput, "\n") { | ||
| line = strings.TrimSpace(line) | ||
| // Ignore comment lines | ||
| if strings.HasPrefix(line, "#") { | ||
| continue | ||
| } | ||
| fields := strings.Fields(line) | ||
| // Ignore lines that don't have at least a metric name and value | ||
| if len(fields) < 2 { | ||
| continue | ||
| } | ||
| metric := fields[0] | ||
|
|
||
| // Extract out the metric class, ignoring any labels. For example, from "DCGM_FI_DEV_GPU_TEMP{gpu=\"0\"}" we want "DCGM_FI_DEV_GPU_TEMP". | ||
| metricClass := metric | ||
| if prefix, _, ok := strings.Cut(metric, "{"); ok { | ||
| metricClass = prefix | ||
| } | ||
|
|
||
| // Record the metric class as found | ||
| metricClasses[metricClass] = true | ||
| } | ||
|
|
||
| allFound := true | ||
| for _, m := range requiredMetrics { | ||
| if !metricClasses[m] { | ||
| allFound = false | ||
| break | ||
| } | ||
| } | ||
| if allFound { | ||
| h.Logf("All required metrics found on attempt %d", attempt) | ||
| break | ||
| } | ||
|
|
||
| if attempt < maxAttempts { | ||
| h.Logf("Attempt %d: not all required metrics found, retrying in 10s...", attempt) | ||
| time.Sleep(10 * time.Second) | ||
| } | ||
| } | ||
|
|
||
| h.Logf("Received metrics:\n%s", metricsOutput) | ||
|
|
||
| for _, m := range requiredMetrics { | ||
| if !metricClasses[m] { | ||
| h.Errorf("Did not find expected metric: %s", m) | ||
| } else { | ||
| h.Logf("Found expected metric: %s", m) | ||
| } | ||
| } | ||
| }) | ||
|
|
||
| if h.AllPassed() { | ||
| h.RecordConformance("observability/accelerator_metrics") | ||
| } | ||
| } | ||
Oops, something went wrong.
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
This retry is why we can't just apply a manifest, BTW. I'll probably extract a helper out for "HTTP GET inside the cluster", but ... starting simple.