Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 18 additions & 12 deletions pkg/collector/corechecks/gpu/nvidia/collector_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -636,22 +636,22 @@ func TestRemoveDuplicateSamples(t *testing.T) {

t.Run("PriorityTie", func(t *testing.T) {
// Edge case: same metric name with same priority across collectors
// First collector (in iteration order) should win
// The collector whose name sorts first should win, regardless of map iteration order
allMetrics := map[CollectorName][]Sample{
sampling: {
metric("metric1", Low, "tagA"),
},
stateless: {
metric("metric1", Low, "tagB"),
},
sampling: {
metric("metric1", Low, "tagA"),
},
}

result := requireMetrics(t, RemoveDuplicateSamples(allMetrics))

// Should have exactly 1 metric (one collector wins the tie)
require.Len(t, result, 1)
require.Equal(t, Low, result[0].Priority())
// Don't assert which specific tag wins since map iteration order is not guaranteed
// Map iteration order is random, so a single call could pick the right collector by chance
for range 100 {
result := requireMetrics(t, RemoveDuplicateSamples(allMetrics))
require.Len(t, result, 1)
require.Equal(t, []string{"tagA"}, result[0].Tags())
}
})

t.Run("EmptyInputs", func(t *testing.T) {
Expand Down Expand Up @@ -775,7 +775,7 @@ func TestConfiguredMetricPriority(t *testing.T) {

// Set up the expected metric order. The first collector in the list should have the highest priority over the rest.
desiredMetricPriority := map[string][]CollectorName{
"sm_active": {sampling, ebpf},
"sm_active": {sampling, ebpf, gpm},
"gr_engine_active": {gpm, sampling, ebpf},
"process.sm_active": {sampling, ebpf},
}
Expand Down Expand Up @@ -829,7 +829,13 @@ func TestConfiguredMetricPriority(t *testing.T) {
for i := range len(collectorOrder) - 1 {
higherPriorityCollector := collectorOrder[i]
lowerPriorityCollector := collectorOrder[i+1]
require.Greater(t, metricMap[higherPriorityCollector].Priority(), metricMap[lowerPriorityCollector].Priority(), "collector %s should have higher priority than collector %s", higherPriorityCollector, lowerPriorityCollector)
higherPriority, lowerPriority := metricMap[higherPriorityCollector].Priority(), metricMap[lowerPriorityCollector].Priority()
if higherPriority == lowerPriority {
// RemoveDuplicateSamples resolves ties in favor of the collector whose name sorts first
require.Less(t, higherPriorityCollector, lowerPriorityCollector, "collector %s should win the tie with collector %s", higherPriorityCollector, lowerPriorityCollector)
continue
}
require.Greater(t, higherPriority, lowerPriority, "collector %s should have higher priority than collector %s", higherPriorityCollector, lowerPriorityCollector)
}
})
}
Expand Down
23 changes: 23 additions & 0 deletions pkg/collector/corechecks/gpu/nvidia/gpm.go
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,9 @@ type gpmCollector struct {
metricsToCollect map[nvml.GpmMetricId]gpmMetric
nextSampleToCollect int
emitLegacySMActive bool
// emitGrEngineSMActive reports GRAPHICS_UTIL (gr_engine_active) as sm_active too, with grEngineSMActivePriority
emitGrEngineSMActive bool
grEngineSMActivePriority MetricPriority
}

type gpmMetric struct {
Expand Down Expand Up @@ -95,6 +98,15 @@ func newGPMCollectorWithMetrics(device ddnvml.Device, metricsToCollect map[nvml.
}
if deps != nil {
collector.emitLegacySMActive = deps.Config.LegacySMActive
// GRAPHICS_UTIL closely tracks the time any SM was active, so it's also reported as sm_active. Not on MIG
// devices or MIG-enabled GPUs, where it hasn't been validated, nor with the legacy sm_active. Low priority by
// default so that it's only a fallback: it loses the tie with ebpf, as the collector whose name sorts first wins.
physicalDevice, isPhysical := device.(*ddnvml.PhysicalDevice)
collector.emitGrEngineSMActive = isPhysical && !physicalDevice.HasMIGFeatureEnabled && !deps.Config.LegacySMActive
collector.grEngineSMActivePriority = Low
if deps.Config.PreferGrEngineSMActive {
collector.grEngineSMActivePriority = High
}
}

if isMig {
Expand Down Expand Up @@ -263,6 +275,9 @@ func (c *gpmCollector) Collect() ([]Sample, error) {
if c.emitLegacySMActive {
metricCapacity++
}
if c.emitGrEngineSMActive {
metricCapacity++
}
samples := make([]Sample, 0, metricCapacity)
var errs []error
for i := uint32(0); i < gpmMetrics.NumMetrics; i++ {
Expand Down Expand Up @@ -292,6 +307,14 @@ func (c *gpmCollector) Collect() ([]Sample, error) {
Type: metricData.metricType,
})
}
if c.emitGrEngineSMActive && nvml.GpmMetricId(metric.MetricId) == nvml.GPM_METRIC_GRAPHICS_UTIL {
samples = append(samples, &Metric{
baseSample: baseSample{priority: c.grEngineSMActivePriority},
Name: "sm_active",
Value: metric.Value,
Type: metricData.metricType,
})
}
}

return samples, errors.Join(errs...)
Expand Down
72 changes: 70 additions & 2 deletions pkg/collector/corechecks/gpu/nvidia/gpm_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@ import (
"github.com/stretchr/testify/require"

gpuconfig "github.com/DataDog/datadog-agent/pkg/gpu/config"
ddnvml "github.com/DataDog/datadog-agent/pkg/gpu/safenvml"
nvmltestutil "github.com/DataDog/datadog-agent/pkg/gpu/safenvml/testutil"
"github.com/DataDog/datadog-agent/pkg/gpu/testutil"
"github.com/DataDog/datadog-agent/pkg/metrics"
Expand Down Expand Up @@ -190,14 +191,14 @@ func TestGPMCollectorCollectReturnsMetrics(t *testing.T) {

result, err := gpmCol.Collect()
assert.NoError(t, err)
assert.Len(t, result, 2)
assert.Len(t, result, 3) // metric1 uses the GRAPHICS_UTIL ID, so it's also reported as sm_active

foundMetrics := make(map[string]bool)
for _, metric := range requireMetrics(t, result) {
foundMetrics[metric.Name] = true

switch metric.Name {
case "metric1":
case "metric1", "sm_active":
assert.Equal(t, 43.0, metric.Value)
case "metric3":
assert.Equal(t, 45.0, metric.Value)
Expand Down Expand Up @@ -269,3 +270,70 @@ func TestGPMCollectorLegacySMActive(t *testing.T) {
})
}
}

func TestGPMCollectorGrEngineSMActive(t *testing.T) {
const grEngineValue, smUtilValue = 61.0, 59.0

for _, tc := range []struct {
name string
config gpuconfig.Config
mig bool
migParent bool
expectedPriority MetricPriority
expectedValue float64
}{
{name: "default", config: gpuconfig.Config{}, expectedPriority: Low, expectedValue: grEngineValue},
{name: "preferred", config: gpuconfig.Config{PreferGrEngineSMActive: true}, expectedPriority: High, expectedValue: grEngineValue},
{
// The legacy sm_active (SM_UTIL) takes precedence, so gr_engine_active is not reported as sm_active.
name: "preferred with legacy",
config: gpuconfig.Config{PreferGrEngineSMActive: true, LegacySMActive: true},
expectedPriority: High,
expectedValue: smUtilValue,
},
{name: "MIG device", config: gpuconfig.Config{PreferGrEngineSMActive: true}, mig: true},
{name: "MIG parent", config: gpuconfig.Config{PreferGrEngineSMActive: true}, migParent: true},
} {
t.Run(tc.name, func(t *testing.T) {
mockLib := nvmltestutil.SetupMockNVML(t,
testutil.WithGpmSupport(true),
testutil.WithGpmMetricValues(map[nvml.GpmMetricId]testutil.MockGpmMetricValue{
nvml.GPM_METRIC_GRAPHICS_UTIL: {Value: grEngineValue, Return: nvml.SUCCESS},
nvml.GPM_METRIC_SM_UTIL: {Value: smUtilValue, Return: nvml.SUCCESS},
}),
)
physicalDevice := nvmltestutil.PhysicalDevice(t, mockLib, 0)
var device ddnvml.Device = physicalDevice
if tc.mig {
device = &ddnvml.MIGDevice{Parent: physicalDevice, MIGInstanceID: 1}
}
if tc.migParent {
// MIG mode enabled, even if no MIG instances have been created yet
physicalDevice.HasMIGFeatureEnabled = true
}

collector, err := newGPMCollectorWithMetrics(device, map[nvml.GpmMetricId]gpmMetric{
nvml.GPM_METRIC_GRAPHICS_UTIL: {name: "gr_engine_active", metricType: metrics.GaugeType},
nvml.GPM_METRIC_SM_UTIL: {name: "sm_utilization", metricType: metrics.GaugeType},
}, &CollectorDependencies{Config: tc.config})
require.NoError(t, err)

samples, err := collector.Collect()
require.NoError(t, err)
var smActive []*Metric
for _, metric := range requireMetrics(t, samples) {
if metric.Name == "sm_active" {
smActive = append(smActive, metric)
}
}

if tc.mig || tc.migParent {
require.Empty(t, smActive)
return
}
require.Len(t, smActive, 1)
assert.Equal(t, tc.expectedPriority, smActive[0].Priority())
assert.Equal(t, tc.expectedValue, smActive[0].Value)
})
}
}
14 changes: 9 additions & 5 deletions pkg/collector/corechecks/gpu/nvidia/helpers.go
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,8 @@ import (
"errors"
"fmt"
"io"
"maps"
"slices"
"time"

"github.com/NVIDIA/go-nvml/pkg/nvml"
Expand Down Expand Up @@ -200,13 +202,15 @@ func RemoveDuplicateSamples(allSamples map[CollectorName][]Sample) []Sample {
var result []Sample

// For each sample key, pick all matching samples from the collector with the highest-priority sample.
// Ties between collectors are resolved in favor of the collector whose name sorts first, so that the
// selected source doesn't change between runs.
for _, collectorSamples := range keyToCollectorSamples {
maxPriority := Low
var winningPriority MetricPriority
var winningSamples []Sample
for _, prioritySamples := range collectorSamples {
for priority, samples := range prioritySamples {
if priority >= maxPriority {
maxPriority = priority
for _, collectorID := range slices.Sorted(maps.Keys(collectorSamples)) {
for priority, samples := range collectorSamples[collectorID] {
if winningSamples == nil || priority > winningPriority {
winningPriority = priority
winningSamples = samples
}
}
Expand Down
10 changes: 10 additions & 0 deletions pkg/config/schema/yaml/gpu.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,16 @@ properties:
Set to true to restore the legacy mapping of GPM SM utilization to
gpu.sm_active, matching the behavior of DCGM. On GPM-capable GPUs, the high-priority GPM value is
reported as both gpu.sm_active and gpu.sm_utilization.
prefer_gr_engine_sm_active:
node_type: setting
type: boolean
default: false
visibility: public
description: |-
On physical NVIDIA GPUs without MIG, gpu.sm_active falls back to the GPM GR engine activity (the
same value as gpu.gr_engine_active) when no other source is available. Set to true to report it
with priority over the sampling and eBPF sources. This source is not used if gpu.legacy_sm_active
is enabled.
nccl:
node_type: section
type: object
Expand Down
7 changes: 7 additions & 0 deletions pkg/gpu/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -248,6 +248,13 @@ Active time is derived from kernel execution intervals captured within each coll
- **Per-process**: Merge intervals for a single process, then compute the percentage of the window that was active.
- **Device-wide**: Merge intervals across all processes on the device, then compute the percentage of the window that was active.

### Other `sm_active` Sources

- `sampling` collector (Medium): derived from NVML per-process utilization samples.
- `gpm` collector with `gpu.legacy_sm_active` (High): `GPM_METRIC_SM_UTIL`.
- `gpm` collector, physical GPUs without MIG (Low, High with `gpu.prefer_gr_engine_sm_active`, off with
`gpu.legacy_sm_active`): `GPM_METRIC_GRAPHICS_UTIL`. It ties with `ebpf` at Low, and `ebpf` wins (name order).

## GPU Spec Guidance

The GPU spec defines what metrics and tags the core check is expected to emit across architectures and device modes.
Expand Down
4 changes: 4 additions & 0 deletions pkg/gpu/config/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,9 @@ type Config struct {
NVLinkFECLightErrorThreshold int
// LegacySMActive indicates whether the legacy sm_active metric should be emitted.
LegacySMActive bool
// PreferGrEngineSMActive indicates whether the sm_active metric reported from the GPM GR engine activity should
// have priority over the other sm_active sources.
PreferGrEngineSMActive bool
// StaticMetricsReportingInterval is the reporting interval for static GPU metrics.
StaticMetricsReportingInterval time.Duration
// Enabled indicates whether the GPU monitoring probe is enabled.
Expand Down Expand Up @@ -165,6 +168,7 @@ func New() *Config {
DisabledCollectors: agentCfg.GetStringSlice("gpu.disabled_collectors"),
NVLinkFECLightErrorThreshold: agentCfg.GetInt("gpu.nvlink.fec_light_error_threshold"),
LegacySMActive: agentCfg.GetBool("gpu.legacy_sm_active"),
PreferGrEngineSMActive: agentCfg.GetBool("gpu.prefer_gr_engine_sm_active"),
StaticMetricsReportingInterval: agentCfg.GetDuration("gpu.static_metrics_reporting_interval"),
ScanProcessesInterval: time.Duration(spCfg.GetInt(sysconfig.FullKeyPath(consts.GPUNS, "process_scan_interval_seconds"))) * time.Second,
InitialProcessSync: spCfg.GetBool(sysconfig.FullKeyPath(consts.GPUNS, "initial_process_sync")),
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
enhancements:
- |
gpum: On physical NVIDIA GPUs without MIG, ``gpu.sm_active`` falls back to the GPM GR engine
activity (the same value as ``gpu.gr_engine_active``) when no other source is available, unless
``gpu.legacy_sm_active`` is enabled. Set ``gpu.prefer_gr_engine_sm_active`` to ``true`` to report
it with priority over the other sources.
Loading