From 3582b3aca38d8d26992797e9ac88cd6438a8b2ba Mon Sep 17 00:00:00 2001 From: mesutoezdil Date: Tue, 6 Oct 2026 11:54:42 +0200 Subject: [PATCH 1/3] fix(amd): round core requests up to whole WGPs The AMD device plugin hands out whole WGPs on RDNA, because the CU mask is applied per WGP, and publishes the granularity as custominfo cuPerWGP. Round the core request by it so the scheduler accounts what is actually taken and does not place a slice the plugin then rejects at admission. Signed-off-by: mesutoezdil --- pkg/device/amd/device.go | 14 ++++++++++++++ pkg/device/amd/device_test.go | 27 +++++++++++++++++++++++++++ 2 files changed, 41 insertions(+) diff --git a/pkg/device/amd/device.go b/pkg/device/amd/device.go index 59902e647e..db6bb77060 100644 --- a/pkg/device/amd/device.go +++ b/pkg/device/amd/device.go @@ -336,6 +336,11 @@ func (amddevice *AMDDevices) Fit(devices []*device.DeviceUsage, request device.C } coreReq = dev.Totalcore * k.Coresreq / 100 coreReq = max(coreReq, 1) + // RDNA applies the CU mask per WGP, so the device plugin hands out + // whole WGPs; account for the same count. + if unit := cuPerWGP(dev.CustomInfo); unit > 1 { + coreReq = (coreReq + unit - 1) / unit * unit + } coreReq = min(coreReq, dev.Totalcore) } else if dev.Totalmem > 0 && memReq >= dev.Totalmem { // Memreq omitted or zero means whole-card memory; treat core request as whole-card as well. @@ -381,3 +386,12 @@ func (amddevice *AMDDevices) Fit(devices []*device.DeviceUsage, request device.C } return false, tmpDevs, common.GenReason(reason, len(devices)) } + +// cuPerWGP returns how many CUs the device plugin allocates together, as +// published in the registration custominfo; 1 when absent. +func cuPerWGP(info map[string]any) int32 { + if v, ok := info["cuPerWGP"].(float64); ok && v > 1 { + return int32(v) + } + return 1 +} diff --git a/pkg/device/amd/device_test.go b/pkg/device/amd/device_test.go index fe1a58b0a1..0916afcec0 100644 --- a/pkg/device/amd/device_test.go +++ b/pkg/device/amd/device_test.go @@ -415,6 +415,33 @@ func TestDevices_Fit(t *testing.T) { assert.Equal(t, "", reason) }) + t.Run("rounds cores up to whole WGPs on RDNA", func(t *testing.T) { + for _, tc := range []struct { + name string + total, used, coresReq int32 + info map[string]any + wantCores int32 + wantOK bool + }{ + {"rdna odd cores", 64, 0, 5, map[string]any{"cuPerWGP": float64(2)}, 4, true}, + {"cdna keeps single cus", 64, 0, 5, map[string]any{}, 3, true}, + {"one-wgp apu", 2, 0, 50, map[string]any{"cuPerWGP": float64(2)}, 2, true}, + {"apu wgp already taken", 2, 2, 50, map[string]any{"cuPerWGP": float64(2)}, 0, false}, + } { + devices := []*device.DeviceUsage{{ + ID: "dev-0", Count: 10, Totalmem: 1000, Totalcore: tc.total, Usedcores: tc.used, + Type: AMDDevice, Health: true, CustomInfo: tc.info, + }} + req := device.ContainerDeviceRequest{Nums: 1, Type: AMDDevice, Memreq: 100, Coresreq: tc.coresReq} + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Annotations: map[string]string{}}} + ok, got, _ := dev.Fit(devices, req, pod, &device.NodeInfo{}, &device.PodDevices{}) + assert.Equal(t, tc.wantOK, ok, tc.name) + if tc.wantOK { + assert.Equal(t, tc.wantCores, got[AMDDevice][0].Usedcores, tc.name) + } + } + }) + t.Run("retains the registered product type in the allocation", func(t *testing.T) { const productType = "AMD_Instinct_MI300X_VF" devices := []*device.DeviceUsage{ From 070101260b7a0f71baac8b0376df167bf0072a52 Mon Sep 17 00:00:00 2001 From: mesutoezdil Date: Wed, 7 Oct 2026 20:44:27 +0200 Subject: [PATCH 2/3] fix(amd): round WGP core requests in int64 to avoid overflow Signed-off-by: mesutoezdil --- pkg/device/amd/device.go | 4 ++-- pkg/device/amd/device_test.go | 1 + 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/pkg/device/amd/device.go b/pkg/device/amd/device.go index db6bb77060..ccd657d3c5 100644 --- a/pkg/device/amd/device.go +++ b/pkg/device/amd/device.go @@ -339,7 +339,7 @@ func (amddevice *AMDDevices) Fit(devices []*device.DeviceUsage, request device.C // RDNA applies the CU mask per WGP, so the device plugin hands out // whole WGPs; account for the same count. if unit := cuPerWGP(dev.CustomInfo); unit > 1 { - coreReq = (coreReq + unit - 1) / unit * unit + coreReq = int32(min((int64(coreReq)+int64(unit)-1)/int64(unit)*int64(unit), int64(dev.Totalcore))) } coreReq = min(coreReq, dev.Totalcore) } else if dev.Totalmem > 0 && memReq >= dev.Totalmem { @@ -390,7 +390,7 @@ func (amddevice *AMDDevices) Fit(devices []*device.DeviceUsage, request device.C // cuPerWGP returns how many CUs the device plugin allocates together, as // published in the registration custominfo; 1 when absent. func cuPerWGP(info map[string]any) int32 { - if v, ok := info["cuPerWGP"].(float64); ok && v > 1 { + if v, ok := info["cuPerWGP"].(float64); ok && v > 1 && v <= math.MaxInt32 { return int32(v) } return 1 diff --git a/pkg/device/amd/device_test.go b/pkg/device/amd/device_test.go index 0916afcec0..9d90075701 100644 --- a/pkg/device/amd/device_test.go +++ b/pkg/device/amd/device_test.go @@ -426,6 +426,7 @@ func TestDevices_Fit(t *testing.T) { {"rdna odd cores", 64, 0, 5, map[string]any{"cuPerWGP": float64(2)}, 4, true}, {"cdna keeps single cus", 64, 0, 5, map[string]any{}, 3, true}, {"one-wgp apu", 2, 0, 50, map[string]any{"cuPerWGP": float64(2)}, 2, true}, + {"huge wgp size does not overflow", 4, 0, 75, map[string]any{"cuPerWGP": float64(math.MaxInt32)}, 4, true}, {"apu wgp already taken", 2, 2, 50, map[string]any{"cuPerWGP": float64(2)}, 0, false}, } { devices := []*device.DeviceUsage{{ From 7cac225626b8699ea8a503b5336dbeab84c19118 Mon Sep 17 00:00:00 2001 From: mesutoezdil Date: Wed, 7 Oct 2026 21:35:12 +0200 Subject: [PATCH 3/3] fix(amd): ignore fractional cuPerWGP values Signed-off-by: mesutoezdil --- pkg/device/amd/device.go | 2 +- pkg/device/amd/device_test.go | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/pkg/device/amd/device.go b/pkg/device/amd/device.go index ccd657d3c5..3db12f5f33 100644 --- a/pkg/device/amd/device.go +++ b/pkg/device/amd/device.go @@ -390,7 +390,7 @@ func (amddevice *AMDDevices) Fit(devices []*device.DeviceUsage, request device.C // cuPerWGP returns how many CUs the device plugin allocates together, as // published in the registration custominfo; 1 when absent. func cuPerWGP(info map[string]any) int32 { - if v, ok := info["cuPerWGP"].(float64); ok && v > 1 && v <= math.MaxInt32 { + if v, ok := info["cuPerWGP"].(float64); ok && v > 1 && v <= math.MaxInt32 && v == math.Trunc(v) { return int32(v) } return 1 diff --git a/pkg/device/amd/device_test.go b/pkg/device/amd/device_test.go index 9d90075701..7cad1e1eda 100644 --- a/pkg/device/amd/device_test.go +++ b/pkg/device/amd/device_test.go @@ -427,6 +427,7 @@ func TestDevices_Fit(t *testing.T) { {"cdna keeps single cus", 64, 0, 5, map[string]any{}, 3, true}, {"one-wgp apu", 2, 0, 50, map[string]any{"cuPerWGP": float64(2)}, 2, true}, {"huge wgp size does not overflow", 4, 0, 75, map[string]any{"cuPerWGP": float64(math.MaxInt32)}, 4, true}, + {"fractional wgp size is ignored", 64, 0, 5, map[string]any{"cuPerWGP": 2.5}, 3, true}, {"apu wgp already taken", 2, 2, 50, map[string]any{"cuPerWGP": float64(2)}, 0, false}, } { devices := []*device.DeviceUsage{{