Skip to content

Commit 0ef6740

Browse files
authored
feat(monitor): report per-container VRAM and host GPU metrics (#150)
Signed-off-by: mesutoezdil <mesudozdil@gmail.com>
1 parent 986f203 commit 0ef6740

14 files changed

Lines changed: 1124 additions & 3 deletions

File tree

‎.github/workflows/ci.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -76,7 +76,7 @@ jobs:
7676
helm template amd-gpu ./helm/amd-gpu --namespace kube-system >/dev/null
7777
# the optional branches render too
7878
helm template amd-gpu ./helm/amd-gpu --namespace kube-system \
79-
--set dp.cdi.enabled=true,dp.muslFailClosed.enabled=true,node_selector_enabled=true >/dev/null
79+
--set dp.cdi.enabled=true,dp.muslFailClosed.enabled=true,node_selector_enabled=true,monitor.enabled=true >/dev/null
8080
8181
# Build-only checks so the images that CI does not publish cannot silently break.
8282
extra-images:

‎Dockerfile‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,7 @@ ADD . /go/src/github.com/Project-HAMi/amd-device-plugin
3434
WORKDIR /go/src/github.com/Project-HAMi/amd-device-plugin/cmd/k8s-device-plugin
3535
RUN go install \
3636
-ldflags="-X main.gitDescribe=$(git -C /go/src/github.com/Project-HAMi/amd-device-plugin/ describe --always --long --dirty 2>/dev/null || echo unknown)"
37+
RUN CGO_ENABLED=0 go install ../k8s-vgpu-monitor
3738

3839
FROM rocm-runtime
3940
LABEL \
@@ -48,6 +49,7 @@ COPY --from=amdsmi-sdk /opt/rocm-7.2.4/share/amd_smi/amdsmi/libamd_smi.so /opt/r
4849
RUN mkdir -p /opt/hami/bin /opt/hami/lib/amd
4950
WORKDIR /root/
5051
COPY --from=builder /go/bin/k8s-device-plugin .
52+
COPY --from=builder /go/bin/k8s-vgpu-monitor .
5153
COPY --from=amdsmi-sdk /build/amd-hami-core/build-hip/libamvgpu.so /opt/hami/lib/amd/libamvgpu.so
5254
COPY --from=builder /go/src/github.com/Project-HAMi/amd-device-plugin/scripts/amd-vgpu-init.sh /opt/hami/bin/amd-vgpu-init.sh
5355
RUN chmod 0555 /opt/hami/bin/amd-vgpu-init.sh \

‎cmd/k8s-vgpu-monitor/main.go‎

Lines changed: 144 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,144 @@
1+
/*
2+
Copyright 2026 The HAMi Authors.
3+
4+
Licensed under the Apache License, Version 2.0 (the "License");
5+
you may not use this file except in compliance with the License.
6+
You may obtain a copy of the License at
7+
8+
http://www.apache.org/licenses/LICENSE-2.0
9+
10+
Unless required by applicable law or agreed to in writing, software
11+
distributed under the License is distributed on an "AS IS" BASIS,
12+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13+
See the License for the specific language governing permissions and
14+
limitations under the License.
15+
*/
16+
17+
// Command k8s-vgpu-monitor serves Prometheus metrics for the AMD GPUs of one
18+
// node: per-container VRAM from the dmem cgroup controller and host memory,
19+
// load, temperature and power from sysfs, under the metric names of the HAMi
20+
// NVIDIA vGPUmonitor.
21+
package main
22+
23+
import (
24+
"context"
25+
"errors"
26+
"flag"
27+
"fmt"
28+
"net"
29+
"net/http"
30+
"os"
31+
"os/signal"
32+
"syscall"
33+
"time"
34+
35+
"github.com/golang/glog"
36+
"github.com/prometheus/client_golang/prometheus"
37+
"github.com/prometheus/client_golang/prometheus/promhttp"
38+
corev1 "k8s.io/api/core/v1"
39+
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
40+
"k8s.io/apimachinery/pkg/labels"
41+
"k8s.io/client-go/informers"
42+
"k8s.io/client-go/kubernetes"
43+
"k8s.io/client-go/rest"
44+
45+
"github.com/Project-HAMi/amd-device-plugin/internal/pkg/utils"
46+
"github.com/Project-HAMi/amd-device-plugin/internal/pkg/vgpumonitor"
47+
)
48+
49+
func main() {
50+
var bind, cgroupRoot, drmRoot string
51+
flag.StringVar(&bind, "metrics_bind_address", ":9394", "TCP address to serve /metrics on")
52+
flag.StringVar(&cgroupRoot, "cgroup_root", vgpumonitor.DefaultCgroupRoot, "cgroup v2 mount point of the host")
53+
flag.StringVar(&drmRoot, "drm_root", vgpumonitor.DefaultDRMRoot, "sysfs DRM class directory of the host")
54+
flag.Parse()
55+
56+
if err := run(bind, cgroupRoot, drmRoot); err != nil {
57+
glog.Errorf("%v", err)
58+
glog.Flush()
59+
os.Exit(1)
60+
}
61+
}
62+
63+
func run(bind, cgroupRoot, drmRoot string) error {
64+
nodeName := os.Getenv(utils.NodeNameEnvName)
65+
if nodeName == "" {
66+
return fmt.Errorf("env %s not set", utils.NodeNameEnvName)
67+
}
68+
config, err := rest.InClusterConfig()
69+
if err != nil {
70+
return fmt.Errorf("failed to load the in-cluster config: %w", err)
71+
}
72+
clientset, err := kubernetes.NewForConfig(config)
73+
if err != nil {
74+
return fmt.Errorf("failed to build clientset: %w", err)
75+
}
76+
77+
ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
78+
defer cancel()
79+
80+
// Bind before the informer starts: a taken port never heals, so fail fast.
81+
listener, err := net.Listen("tcp", bind)
82+
if err != nil {
83+
return fmt.Errorf("failed to listen on %s: %w", bind, err)
84+
}
85+
defer func() { _ = listener.Close() }()
86+
return serve(ctx, clientset, nodeName, listener, cgroupRoot, drmRoot)
87+
}
88+
89+
// serve runs the collector for nodeName until ctx is done.
90+
func serve(ctx context.Context, clientset kubernetes.Interface, nodeName string, listener net.Listener, cgroupRoot, drmRoot string) error {
91+
factory := informers.NewSharedInformerFactoryWithOptions(clientset, 5*time.Minute,
92+
informers.WithTweakListOptions(func(o *metav1.ListOptions) { o.FieldSelector = "spec.nodeName=" + nodeName }))
93+
podInformer := factory.Core().V1().Pods()
94+
podLister := podInformer.Lister()
95+
synced := podInformer.Informer().HasSynced
96+
factory.Start(ctx.Done())
97+
if !waitForSync(ctx, synced) {
98+
return errors.New("failed to sync pod informer cache")
99+
}
100+
101+
reg := prometheus.NewRegistry()
102+
reg.MustRegister(&vgpumonitor.Collector{
103+
NodeName: nodeName,
104+
CgroupRoot: cgroupRoot,
105+
DRMRoot: drmRoot,
106+
Pods: func() ([]*corev1.Pod, error) {
107+
return podLister.List(labels.Everything())
108+
},
109+
Node: func() (*corev1.Node, error) {
110+
return clientset.CoreV1().Nodes().Get(ctx, nodeName, metav1.GetOptions{})
111+
},
112+
})
113+
glog.Infof("Serving AMD metrics for node %s on %s", nodeName, listener.Addr())
114+
return serveMetrics(ctx, listener, reg)
115+
}
116+
117+
func waitForSync(ctx context.Context, synced func() bool) bool {
118+
for !synced() {
119+
select {
120+
case <-ctx.Done():
121+
return false
122+
case <-time.After(100 * time.Millisecond):
123+
}
124+
}
125+
return true
126+
}
127+
128+
// serveMetrics serves reg on listener until ctx is done or the server fails.
129+
func serveMetrics(ctx context.Context, listener net.Listener, reg *prometheus.Registry) error {
130+
mux := http.NewServeMux()
131+
mux.Handle("/metrics", promhttp.HandlerFor(reg, promhttp.HandlerOpts{}))
132+
server := &http.Server{Handler: mux, ReadHeaderTimeout: 15 * time.Second, ReadTimeout: 60 * time.Second}
133+
134+
errCh := make(chan error, 1)
135+
go func() { errCh <- server.Serve(listener) }()
136+
select {
137+
case err := <-errCh:
138+
return err
139+
case <-ctx.Done():
140+
shutdownCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
141+
defer cancel()
142+
return server.Shutdown(shutdownCtx)
143+
}
144+
}

‎cmd/k8s-vgpu-monitor/main_test.go‎

Lines changed: 121 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,121 @@
1+
/*
2+
Copyright 2026 The HAMi Authors.
3+
4+
Licensed under the Apache License, Version 2.0 (the "License");
5+
you may not use this file except in compliance with the License.
6+
You may obtain a copy of the License at
7+
8+
http://www.apache.org/licenses/LICENSE-2.0
9+
10+
Unless required by applicable law or agreed to in writing, software
11+
distributed under the License is distributed on an "AS IS" BASIS,
12+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13+
See the License for the specific language governing permissions and
14+
limitations under the License.
15+
*/
16+
17+
package main
18+
19+
import (
20+
"context"
21+
"io"
22+
"net"
23+
"net/http"
24+
"os"
25+
"path/filepath"
26+
"testing"
27+
"time"
28+
29+
"github.com/prometheus/client_golang/prometheus"
30+
"github.com/stretchr/testify/require"
31+
corev1 "k8s.io/api/core/v1"
32+
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
33+
"k8s.io/client-go/kubernetes/fake"
34+
35+
"github.com/Project-HAMi/amd-device-plugin/internal/pkg/utils"
36+
)
37+
38+
func TestRunNeedsTheNodeName(t *testing.T) {
39+
t.Setenv(utils.NodeNameEnvName, "")
40+
require.ErrorContains(t, run(":0", t.TempDir(), t.TempDir()), utils.NodeNameEnvName)
41+
}
42+
43+
func TestRunNeedsAnInClusterConfig(t *testing.T) {
44+
t.Setenv(utils.NodeNameEnvName, "gpu-1")
45+
t.Setenv("KUBERNETES_SERVICE_HOST", "")
46+
require.ErrorContains(t, run(":0", t.TempDir(), t.TempDir()), "in-cluster")
47+
}
48+
49+
// serveMetrics must serve /metrics and stop when
50+
// its context ends.
51+
func TestServeMetricsServesAndStops(t *testing.T) {
52+
listener, err := net.Listen("tcp", "127.0.0.1:0")
53+
require.NoError(t, err)
54+
55+
reg := prometheus.NewRegistry()
56+
gauge := prometheus.NewGauge(prometheus.GaugeOpts{Name: "amd_test_gauge", Help: "h"})
57+
gauge.Set(7)
58+
reg.MustRegister(gauge)
59+
60+
ctx, cancel := context.WithCancel(context.Background())
61+
done := make(chan error, 1)
62+
go func() { done <- serveMetrics(ctx, listener, reg) }()
63+
64+
resp, err := http.Get("http://" + listener.Addr().String() + "/metrics")
65+
require.NoError(t, err)
66+
body, _ := io.ReadAll(resp.Body)
67+
resp.Body.Close()
68+
require.Contains(t, string(body), "amd_test_gauge 7")
69+
70+
cancel()
71+
select {
72+
case err := <-done:
73+
require.NoError(t, err)
74+
case <-time.After(5 * time.Second):
75+
t.Fatal("serveMetrics did not stop after its context ended")
76+
}
77+
}
78+
79+
// serve must wire the node's pods and registration into the collector: the
80+
// device memory of a card the node registered shows up on /metrics.
81+
func TestServeAMDServesTheRegisteredCard(t *testing.T) {
82+
const bdf = "0000:06:00.0"
83+
drm := t.TempDir()
84+
card := filepath.Join(drm, "card1", "device")
85+
require.NoError(t, os.MkdirAll(card, 0o755))
86+
for name, content := range map[string]string{
87+
"uevent": "DRIVER=amdgpu\nPCI_SLOT_NAME=" + bdf + "\n",
88+
"mem_info_vram_used": "1024\n",
89+
"gpu_busy_percent": "5\n",
90+
} {
91+
require.NoError(t, os.WriteFile(filepath.Join(card, name), []byte(content), 0o644))
92+
}
93+
clientset := fake.NewSimpleClientset(&corev1.Node{ObjectMeta: metav1.ObjectMeta{
94+
Name: "gpu-1",
95+
Annotations: map[string]string{
96+
"hami.io/node-amd-register": `[{"id":"uuid-1","index":0,"type":"AMD","custominfo":{"pciBDF":"` + bdf + `"}}]`,
97+
},
98+
}})
99+
100+
listener, err := net.Listen("tcp", "127.0.0.1:0")
101+
require.NoError(t, err)
102+
103+
ctx, cancel := context.WithCancel(context.Background())
104+
done := make(chan error, 1)
105+
go func() { done <- serve(ctx, clientset, "gpu-1", listener, t.TempDir(), drm) }()
106+
107+
resp, err := http.Get("http://" + listener.Addr().String() + "/metrics")
108+
require.NoError(t, err)
109+
body, _ := io.ReadAll(resp.Body)
110+
resp.Body.Close()
111+
require.Contains(t, string(body), `hami_host_gpu_memory_used_bytes{device_index="0",device_type="AMD",device_uuid="uuid-1",node="gpu-1"} 1024`)
112+
require.Contains(t, string(body), `hami_vgpumonitor_collect_success{node="gpu-1"} 1`)
113+
114+
cancel()
115+
select {
116+
case err := <-done:
117+
require.NoError(t, err)
118+
case <-time.After(5 * time.Second):
119+
t.Fatal("serve did not stop after its context ended")
120+
}
121+
}

‎docs/user-guide/installation.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -33,6 +33,7 @@ kubectl get node <node> -o jsonpath='{.metadata.annotations.hami\.io/node-amd-re
3333
## Optional components
3434

3535
- **GPU health from the AMD Device Metrics Exporter.** Install the [exporter](https://github.com/ROCm/device-metrics-exporter) with its gRPC socket enabled at `/var/lib/amd-metrics-exporter/`. The plugin then also uses the per-GPU health the exporter reports, for example after ECC errors. Without it, the plugin still checks that each GPU's device node can be opened.
36+
- **GPU metrics.** Set `monitor.enabled=true` in the Helm chart to run `k8s-vgpu-monitor` on each GPU node. It serves Prometheus metrics on `:9394` under the metric names of the HAMi NVIDIA vGPUmonitor: `hami_vgpu_memory_used_bytes` and `hami_vgpu_memory_limit_bytes` per container, and `hami_host_gpu_memory_used_bytes`, `hami_host_gpu_utilization_ratio`, `hami_host_gpu_temperature_celsius` and `hami_host_gpu_power_usage_watts` per GPU. Container memory comes from the dmem cgroup controller, so it needs the systemd cgroup driver and a kernel with the controller. Per-container utilization is not reported, because the kernel does not account compute time per process.
3637
- **Node labeller.** `k8s-ds-amdgpu-labeller.yaml` deploys the upstream labeller, which adds `amd.com/gpu.*` node labels such as VRAM, CU count, device ID and family:
3738

3839
```bash

‎go.mod‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,6 +6,8 @@ require (
66
github.com/go-logr/logr v1.4.4
77
github.com/golang/glog v1.2.5
88
github.com/kubevirt/device-plugin-manager v1.19.5
9+
github.com/prometheus/client_golang v1.24.0
10+
github.com/stretchr/testify v1.11.1
911
google.golang.org/grpc v1.86.0-dev
1012
google.golang.org/protobuf v1.36.12
1113
k8s.io/api v0.37.1
@@ -42,11 +44,11 @@ require (
4244
github.com/google/gnostic-models v0.7.0 // indirect
4345
github.com/google/uuid v1.6.0 // indirect
4446
github.com/json-iterator/go v1.1.12 // indirect
47+
github.com/kylelemons/godebug v1.1.0 // indirect
4548
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
4649
github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee // indirect
4750
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
4851
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect
49-
github.com/prometheus/client_golang v1.24.0 // indirect
5052
github.com/prometheus/client_model v0.6.2 // indirect
5153
github.com/prometheus/common v0.70.0 // indirect
5254
github.com/prometheus/procfs v0.21.1 // indirect
@@ -67,6 +69,7 @@ require (
6769
google.golang.org/genproto/googleapis/rpc v0.0.0-20260817212433-ac3dfec99bb1 // indirect
6870
gopkg.in/evanphx/json-patch.v4 v4.13.0 // indirect
6971
gopkg.in/inf.v0 v0.9.1 // indirect
72+
gopkg.in/yaml.v3 v3.0.1 // indirect
7073
k8s.io/apiextensions-apiserver v0.37.0 // indirect
7174
k8s.io/kube-openapi v0.0.0-20260721132016-d427ff9ee9ad // indirect
7275
k8s.io/utils v0.0.0-20260626114624-be93311217bd // indirect

‎helm/amd-gpu/README.md‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -42,6 +42,9 @@ helm install amd-gpu ./helm/amd-gpu -n kube-system
4242
| dp.cdi.specDir | string | `"/var/run/cdi"` | Host directory the CDI spec is written to; the runtime must read it. |
4343
| dp.resources | object | `{}` | Plugin container resources. |
4444
| dp.updateStrategy | object | `RollingUpdate, maxUnavailable: 1` | DaemonSet update strategy. |
45+
| monitor.enabled | bool | `false` | Run the metrics DaemonSet: per-container VRAM from the dmem cgroup controller and host memory, load, temperature and power from sysfs, under the metric names of the HAMi NVIDIA vGPUmonitor. Needs the systemd cgroup driver on a kernel with the dmem controller. |
46+
| monitor.metricsBindAddress | string | `":9394"` | Address the monitor serves `/metrics` on; the container port and the scrape annotation follow it. |
47+
| monitor.resources | object | `{}` | Monitor container resources. |
4548
| imagePullSecrets | list | `[]` | Image pull secrets. |
4649
| tolerations | list | `[{key: CriticalAddonsOnly, operator: Exists}]` | DaemonSet tolerations. |
4750
| node_selector_enabled | bool | `false` | Restrict the DaemonSet to `node_selector`. |

0 commit comments

Comments
 (0)