diff --git a/cmd/vGPUmonitor/main.go b/cmd/vGPUmonitor/main.go index 1782fedb37..4f5c8c4e5b 100644 --- a/cmd/vGPUmonitor/main.go +++ b/cmd/vGPUmonitor/main.go @@ -142,10 +142,7 @@ func initMetrics(ctx context.Context, containerLister *nvidia.ContainerLister) e reg.MustRegister(versionmetrics.NewBuildInfoCollector()) - // Construct cluster managers. In real code, we would assign them to - // variables to then do something with them. NewClusterManager("vGPU", reg, containerLister, legacyMetrics) - //NewClusterManager("ca", reg) // Uncomment to add the standard process and Go metrics to the custom registry. //reg.MustRegister( diff --git a/cmd/vGPUmonitor/metrics.go b/cmd/vGPUmonitor/metrics.go index ea4b4bf0b4..ae07af2cdc 100644 --- a/cmd/vGPUmonitor/metrics.go +++ b/cmd/vGPUmonitor/metrics.go @@ -38,44 +38,17 @@ import ( "k8s.io/klog/v2" ) -// ClusterManager is an example for a system that might have been built without -// Prometheus in mind. It models a central manager of jobs running in a -// cluster. Thus, we implement a custom Collector called -// ClusterManagerCollector, which collects information from a ClusterManager -// using its provided methods and turns them into Prometheus Metrics for -// collection. -// -// An additional challenge is that multiple instances of the ClusterManager are -// run within the same binary, each in charge of a different zone. We need to -// make use of wrapping Registerers to be able to register each -// ClusterManagerCollector instance with Prometheus. +// ClusterManager holds the state the vGPUMonitor's Prometheus collector reads +// from: the pod and container listers used to gather per-container GPU usage, +// and the zone label each ClusterManagerCollector is registered under via a +// wrapping Registerer. type ClusterManager struct { - Zone string - // Contains many more fields not listed in this example. + Zone string PodLister listerscorev1.PodLister containerLister *nvidia.ContainerLister LegacyMetrics bool } -// ReallyExpensiveAssessmentOfTheSystemState is a mock for the data gathering a -// real cluster manager would have to do. Since it may actually be really -// expensive, it must only be called once per collection. This implementation, -// obviously, only returns some made-up data. -func (c *ClusterManager) ReallyExpensiveAssessmentOfTheSystemState() ( - oomCountByHost map[string]int, ramUsageByHost map[string]float64, -) { - // Just example fake data. - oomCountByHost = map[string]int{ - "foo.example.org": 42, - "bar.example.org": 2001, - } - ramUsageByHost = map[string]float64{ - "foo.example.org": 6.023e23, - "bar.example.org": 3.14, - } - return -} - // ClusterManagerCollector implements the Collector interface. type ClusterManagerCollector struct { ClusterManager *ClusterManager @@ -237,52 +210,9 @@ func (cc ClusterManagerCollector) Describe(ch chan<- *prometheus.Desc) { } } -//func parseidstr(podusage string) (string, string, error) { -// tmp := strings.Split(podusage, "_") -// if len(tmp) > 1 { -// return tmp[0], tmp[1], nil -// } else { -// return "", "", errors.New("parse error") -// } -//} -// -//func gettotalusage(usage podusage, vidx int) (deviceMemory, error) { -// added := deviceMemory{ -// bufferSize: 0, -// contextSize: 0, -// moduleSize: 0, -// offset: 0, -// total: 0, -// } -// for _, val := range usage.sr.procs { -// added.bufferSize += val.used[vidx].bufferSize -// added.contextSize += val.used[vidx].contextSize -// added.moduleSize += val.used[vidx].moduleSize -// added.offset += val.used[vidx].offset -// added.total += val.used[vidx].total -// } -// return added, nil -//} -// -//func getTotalUtilization(usage podusage, vidx int) deviceUtilization { -// added := deviceUtilization{ -// decUtil: 0, -// encUtil: 0, -// smUtil: 0, -// } -// for _, val := range usage.sr.procs { -// added.decUtil += val.deviceUtil[vidx].decUtil -// added.encUtil += val.deviceUtil[vidx].encUtil -// added.smUtil += val.deviceUtil[vidx].smUtil -// } -// return added -//} - -// Collect first triggers the ReallyExpensiveAssessmentOfTheSystemState. Then it -// creates constant metrics for each host on the fly based on the returned data. -// -// Note that Collect could be called concurrently, so we depend on -// ReallyExpensiveAssessmentOfTheSystemState to be concurrency-safe. +// Collect gathers GPU, pod, and container metrics on each scrape and sends them +// to the provided channel. It may be called concurrently, so the collection +// helpers it calls must be concurrency-safe. func (cc ClusterManagerCollector) Collect(ch chan<- prometheus.Metric) { klog.Info("Starting to collect metrics for vGPUMonitor") @@ -621,11 +551,9 @@ func sendMetric(ch chan<- prometheus.Metric, desc *prometheus.Desc, valueType pr return nil } -// NewClusterManager first creates a Prometheus-ignorant ClusterManager -// instance. Then, it creates a ClusterManagerCollector for the just created -// ClusterManager. Finally, it registers the ClusterManagerCollector with a -// wrapping Registerer that adds the zone as a label. In this way, the metrics -// collected by different ClusterManagerCollectors do not collide. +// NewClusterManager creates a ClusterManager for the given zone, backs its pod +// lookups with a shared informer, and registers its collector with reg through +// a wrapping Registerer that adds the zone as a label. func NewClusterManager(zone string, reg prometheus.Registerer, containerLister *nvidia.ContainerLister, legacyMetrics bool) *ClusterManager { if legacyMetrics { initLegacyDescriptors()