@cryptotaxi247 / netdata-1 / commits / e448530a3

nvidia_smi: init version added (#4589)

* nvidia_smi: init version added * Update nvidia_smi.chart.py * nvidia_smi: don't use __bool__ for checking xml parse result

Ilya Mashchenko committed Nov 12, 2018 at 23:47 UTC e448530a3e712f2b503ddc9c0d6ebb9cd7426dd0
6 files changed +482
collectors/python.d.plugin/Makefile.am
+1
@@ -74,6 +74,7 @@ include monit/Makefile.inc
74 include mysql/Makefile.inc
75 include nginx/Makefile.inc
76 include nginx_plus/Makefile.inc
77 +include nvidia_smi/Makefile.inc
78 include nsd/Makefile.inc
79 include ntpd/Makefile.inc
80 include ovpn_status_log/Makefile.inc
collectors/python.d.plugin/nvidia_smi/Makefile.inc new
+12
@@ -0,0 +1,12 @@
1 +# SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +# THIS IS NOT A COMPLETE Makefile
4 +# IT IS INCLUDED BY ITS PARENT'S Makefile.am
5 +# IT IS REQUIRED TO REFERENCE ALL FILES RELATIVE TO THE PARENT
6 +
7 +# install these files
8 +dist_python_DATA += nvidia_smi/nvidia_smi.chart.py
9 +dist_pythonconfig_DATA += nvidia_smi/nvidia_smi.conf
10 +
11 +# do not install these files, but include them in the distribution
12 +dist_noinst_DATA += nvidia_smi/README.md nvidia_smi/Makefile.inc
collectors/python.d.plugin/nvidia_smi/README.md new
+39
@@ -0,0 +1,39 @@
1 +# nvidia_smi
2 +
3 +This module monitors the `nvidia-smi` cli tool.
4 +
5 +**Requirements and Notes:**
6 +
7 + * You must have the `nvidia-smi` tool installed and your NVIDIA GPU(s) must support the tool. Mostly the newer high end models used for AI / ML and Crypto or Pro range, read more about [nvidia_smi](https://developer.nvidia.com/nvidia-system-management-interface).
8 +
9 + * You must enable this plugin as its disabled by default due to minor performance issues.
10 +
11 + * On some systems when the GPU is idle the `nvidia-smi` tool unloads and there is added latency again when it is next queried. If you are running GPUs under constant workload this isn't likely to be an issue.
12 +
13 + * Currently the `nvidia-smi` tool is being queried via cli. Updating the plugin to use the nvidia c/c++ API directly should resolve this issue. See discussion here: https://github.com/netdata/netdata/pull/4357
14 +
15 + * Contributions are welcome.
16 +
17 + * Make sure `netdata` user can execute `/usr/bin/nvidia-smi` or wherever your binary is.
18 +
19 + * `poll_seconds` is how often in seconds the tool is polled for as an integer.
20 +
21 +It produces:
22 +
23 +1. Per GPU
24 + * GPU utilization
25 + * memory allocation
26 + * memory utilization
27 + * fan speed
28 + * power usage
29 + * temperature
30 + * clock speed
31 + * PCI bandwidth
32 +
33 +### configuration
34 +
35 +Sample:
36 +
37 +```yaml
38 +poll_seconds: 1
39 +```
\ No newline at end of file
collectors/python.d.plugin/nvidia_smi/nvidia_smi.chart.py new
+361
@@ -0,0 +1,361 @@
1 +# -*- coding: utf-8 -*-
2 +# Description: nvidia-smi netdata python.d module
3 +# Original Author: Steven Noonan (tycho)
4 +# Author: Ilya Mashchenko (l2isbad)
5 +
6 +import subprocess
7 +import threading
8 +import xml.etree.ElementTree as et
9 +
10 +from bases.collection import find_binary
11 +from bases.FrameworkServices.SimpleService import SimpleService
12 +
13 +disabled_by_default = True
14 +
15 +
16 +NVIDIA_SMI = 'nvidia-smi'
17 +
18 +EMPTY_ROW = ''
19 +EMPTY_ROW_LIMIT = 500
20 +POLLER_BREAK_ROW = '</nvidia_smi_log>'
21 +
22 +PCI_BANDWIDTH = 'pci_bandwidth'
23 +FAN_SPEED = 'fan_speed'
24 +GPU_UTIL = 'gpu_utilization'
25 +MEM_UTIL = 'mem_utilization'
26 +ENCODER_UTIL = 'encoder_utilization'
27 +MEM_ALLOCATED = 'mem_allocated'
28 +TEMPERATURE = 'temperature'
29 +CLOCKS = 'clocks'
30 +POWER = 'power'
31 +
32 +ORDER = [
33 + PCI_BANDWIDTH,
34 + FAN_SPEED,
35 + GPU_UTIL,
36 + MEM_UTIL,
37 + ENCODER_UTIL,
38 + MEM_ALLOCATED,
39 + TEMPERATURE,
40 + CLOCKS,
41 + POWER,
42 +]
43 +
44 +
45 +def gpu_charts(gpu):
46 + fam = gpu.full_name()
47 +
48 + charts = {
49 + PCI_BANDWIDTH: {
50 + 'options': [None, 'PCI Express Bandwidth Utilization', 'KB/s', fam, 'nvidia_smi.pci_bandwidth', 'area'],
51 + 'lines': [
52 + ['rx_util', 'rx', 'absolute', 1, 1],
53 + ['tx_util', 'tx', 'absolute', 1, -1],
54 + ]
55 + },
56 + FAN_SPEED: {
57 + 'options': [None, 'Fan Speed', '%', fam, 'nvidia_smi.fan_speed', 'line'],
58 + 'lines': [
59 + ['fan_speed', 'speed'],
60 + ]
61 + },
62 + GPU_UTIL: {
63 + 'options': [None, 'GPU Utilization', '%', fam, 'nvidia_smi.gpu_utilization', 'line'],
64 + 'lines': [
65 + ['gpu_util', 'utilization'],
66 + ]
67 + },
68 + MEM_UTIL: {
69 + 'options': [None, 'Memory Bandwidth Utilization', '%', fam, 'nvidia_smi.mem_utilization', 'line'],
70 + 'lines': [
71 + ['memory_util', 'utilization'],
72 + ]
73 + },
74 + ENCODER_UTIL: {
75 + 'options': [None, 'Encoder/Decoder Utilization', '%', fam, 'nvidia_smi.encoder_utilization', 'line'],
76 + 'lines': [
77 + ['encoder_util', 'encoder'],
78 + ['decoder_util', 'decoder'],
79 + ]
80 + },
81 + MEM_ALLOCATED: {
82 + 'options': [None, 'Memory Allocated', 'MB', fam, 'nvidia_smi.memory_allocated', 'line'],
83 + 'lines': [
84 + ['fb_memory_usage', 'used'],
85 + ]
86 + },
87 + TEMPERATURE: {
88 + 'options': [None, 'Temperature', 'celsius', fam, 'nvidia_smi.temperature', 'line'],
89 + 'lines': [
90 + ['gpu_temp', 'temp'],
91 + ]
92 + },
93 + CLOCKS: {
94 + 'options': [None, 'Clock Frequencies', 'MHz', fam, 'nvidia_smi.clocks', 'line'],
95 + 'lines': [
96 + ['graphics_clock', 'graphics'],
97 + ['video_clock', 'video'],
98 + ['sm_clock', 'sm'],
99 + ['mem_clock', 'mem'],
100 + ]
101 + },
102 + POWER: {
103 + 'options': [None, 'Power Utilization', 'Watts', fam, 'nvidia_smi.power', 'line'],
104 + 'lines': [
105 + ['power_draw', 'power', 1, 100],
106 + ]
107 + },
108 + }
109 +
110 + idx = gpu.num
111 +
112 + order = ['gpu{0}_{1}'.format(idx, v) for v in ORDER]
113 + charts = dict(('gpu{0}_{1}'.format(idx, k), v) for k, v in charts.items())
114 +
115 + for chart in charts.values():
116 + for line in chart['lines']:
117 + line[0] = 'gpu{0}_{1}'.format(idx, line[0])
118 +
119 + return order, charts
120 +
121 +
122 +class NvidiaSMI:
123 + def __init__(self):
124 + self.command = find_binary(NVIDIA_SMI)
125 + self.active_proc = None
126 +
127 + def run_once(self):
128 + proc = subprocess.Popen([self.command, '-x', '-q'], stdout=subprocess.PIPE)
129 + stdout, _ = proc.communicate()
130 + return stdout
131 +
132 + def run_loop(self, interval):
133 + if self.active_proc:
134 + self.kill()
135 + proc = subprocess.Popen([self.command, '-x', '-q', '-l', str(interval)], stdout=subprocess.PIPE)
136 + self.active_proc = proc
137 + return proc.stdout
138 +
139 + def kill(self):
140 + if self.active_proc:
141 + self.active_proc.kill()
142 + self.active_proc = None
143 +
144 +
145 +class NvidiaSMIPoller(threading.Thread):
146 + def __init__(self, poll_interval):
147 + threading.Thread.__init__(self)
148 + self.daemon = True
149 +
150 + self.smi = NvidiaSMI()
151 + self.interval = poll_interval
152 +
153 + self.lock = threading.RLock()
154 + self.last_data = str()
155 + self.exit = False
156 + self.empty_rows = 0
157 + self.rows = list()
158 +
159 + def has_smi(self):
160 + return bool(self.smi.command)
161 +
162 + def run_once(self):
163 + return self.smi.run_once()
164 +
165 + def run(self):
166 + out = self.smi.run_loop(self.interval)
167 +
168 + for row in out:
169 + if self.exit or self.empty_rows > EMPTY_ROW_LIMIT:
170 + break
171 + self.process_row(row)
172 + self.smi.kill()
173 +
174 + def process_row(self, row):
175 + row = row.decode()
176 + self.empty_rows += (row == EMPTY_ROW)
177 + self.rows.append(row)
178 +
179 + if POLLER_BREAK_ROW in row:
180 + self.lock.acquire()
181 + self.last_data = '\n'.join(self.rows)
182 + self.lock.release()
183 +
184 + self.rows = list()
185 + self.empty_rows = 0
186 +
187 + def is_started(self):
188 + return self.ident is not None
189 +
190 + def shutdown(self):
191 + self.exit = True
192 +
193 + def data(self):
194 + self.lock.acquire()
195 + data = self.last_data
196 + self.lock.release()
197 + return data
198 +
199 +
200 +def handle_attr_error(method):
201 + def on_call(*args, **kwargs):
202 + try:
203 + return method(*args, **kwargs)
204 + except AttributeError:
205 + return None
206 + return on_call
207 +
208 +
209 +class GPU:
210 + def __init__(self, num, root):
211 + self.num = num
212 + self.root = root
213 +
214 + def id(self):
215 + return self.root.get('id')
216 +
217 + def name(self):
218 + return self.root.find('product_name').text
219 +
220 + def full_name(self):
221 + return 'gpu{0} {1}'.format(self.num, self.name())
222 +
223 + @handle_attr_error
224 + def rx_util(self):
225 + return self.root.find('pci').find('rx_util').text.split()[0]
226 +
227 + @handle_attr_error
228 + def tx_util(self):
229 + return self.root.find('pci').find('tx_util').text.split()[0]
230 +
231 + @handle_attr_error
232 + def fan_speed(self):
233 + return self.root.find('fan_speed').text.split()[0]
234 +
235 + @handle_attr_error
236 + def gpu_util(self):
237 + return self.root.find('utilization').find('gpu_util').text.split()[0]
238 +
239 + @handle_attr_error
240 + def memory_util(self):
241 + return self.root.find('utilization').find('memory_util').text.split()[0]
242 +
243 + @handle_attr_error
244 + def encoder_util(self):
245 + return self.root.find('utilization').find('encoder_util').text.split()[0]
246 +
247 + @handle_attr_error
248 + def decoder_util(self):
249 + return self.root.find('utilization').find('decoder_util').text.split()[0]
250 +
251 + @handle_attr_error
252 + def fb_memory_usage(self):
253 + return self.root.find('fb_memory_usage').find('used').text.split()[0]
254 +
255 + @handle_attr_error
256 + def temperature(self):
257 + return self.root.find('temperature').find('gpu_temp').text.split()[0]
258 +
259 + @handle_attr_error
260 + def graphics_clock(self):
261 + return self.root.find('clocks').find('graphics_clock').text.split()[0]
262 +
263 + @handle_attr_error
264 + def video_clock(self):
265 + return self.root.find('clocks').find('video_clock').text.split()[0]
266 +
267 + @handle_attr_error
268 + def sm_clock(self):
269 + return self.root.find('clocks').find('sm_clock').text.split()[0]
270 +
271 + @handle_attr_error
272 + def mem_clock(self):
273 + return self.root.find('clocks').find('mem_clock').text.split()[0]
274 +
275 + @handle_attr_error
276 + def power_draw(self):
277 + return float(self.root.find('power_readings').find('power_draw').text.split()[0]) * 100
278 +
279 + def data(self):
280 + data = {
281 + 'rx_util': self.rx_util(),
282 + 'tx_util': self.tx_util(),
283 + 'fan_speed': self.fan_speed(),
284 + 'gpu_util': self.gpu_util(),
285 + 'memory_util': self.memory_util(),
286 + 'encoder_util': self.encoder_util(),
287 + 'decoder_util': self.decoder_util(),
288 + 'fb_memory_usage': self.fb_memory_usage(),
289 + 'gpu_temp': self.temperature(),
290 + 'graphics_clock': self.graphics_clock(),
291 + 'video_clock': self.video_clock(),
292 + 'sm_clock': self.sm_clock(),
293 + 'mem_clock': self.mem_clock(),
294 + 'power_draw': self.power_draw(),
295 + }
296 +
297 + return dict(('gpu{0}_{1}'.format(self.num, k), v) for k, v in data.items() if v is not None)
298 +
299 +
300 +class Service(SimpleService):
301 + def __init__(self, configuration=None, name=None):
302 + super(Service, self).__init__(configuration=configuration, name=name)
303 + self.order = list()
304 + self.definitions = dict()
305 +
306 + poll = int(configuration.get('poll_seconds', 1))
307 + self.poller = NvidiaSMIPoller(poll)
308 +
309 + def get_data(self):
310 + if not self.poller.is_alive():
311 + self.debug('poller is off')
312 + return None
313 +
314 + last_data = self.poller.data()
315 +
316 + parsed = self.parse_xml(last_data)
317 + if parsed is None:
318 + return None
319 +
320 + data = dict()
321 + for idx, root in enumerate(parsed.findall('gpu')):
322 + data.update(GPU(idx, root).data())
323 +
324 + return data or None
325 +
326 + def check(self):
327 + if not self.poller.has_smi():
328 + self.error("couldn't find '{0}' binary".format(NVIDIA_SMI))
329 + return False
330 +
331 + raw_data = self.poller.run_once()
332 + if not raw_data:
333 + self.error("failed to invoke '{0}' binary".format(NVIDIA_SMI))
334 + return False
335 +
336 + parsed = self.parse_xml(raw_data)
337 + if parsed is None:
338 + return False
339 +
340 + gpus = parsed.findall('gpu')
341 + if not gpus:
342 + return False
343 +
344 + self.create_charts(gpus)
345 + self.poller.start()
346 +
347 + return True
348 +
349 + def parse_xml(self, data):
350 + try:
351 + return et.fromstring(data)
352 + except et.ParseError as error:
353 + self.error(error)
354 +
355 + return None
356 +
357 + def create_charts(self, gpus):
358 + for idx, root in enumerate(gpus):
359 + order, charts = gpu_charts(GPU(idx, root))
360 + self.order.extend(order)
361 + self.definitions.update(charts)
collectors/python.d.plugin/nvidia_smi/nvidia_smi.conf new
+68
@@ -0,0 +1,68 @@
1 +# netdata python.d.plugin configuration for nvidia_smi
2 +#
3 +# This file is in YaML format. Generally the format is:
4 +#
5 +# name: value
6 +#
7 +# There are 2 sections:
8 +# - global variables
9 +# - one or more JOBS
10 +#
11 +# JOBS allow you to collect values from multiple sources.
12 +# Each source will have its own set of charts.
13 +#
14 +# JOB parameters have to be indented (using spaces only, example below).
15 +
16 +# ----------------------------------------------------------------------
17 +# Global Variables
18 +# These variables set the defaults for all JOBs, however each JOB
19 +# may define its own, overriding the defaults.
20 +
21 +# update_every sets the default data collection frequency.
22 +# If unset, the python.d.plugin default is used.
23 +# update_every: 1
24 +
25 +# priority controls the order of charts at the netdata dashboard.
26 +# Lower numbers move the charts towards the top of the page.
27 +# If unset, the default for python.d.plugin is used.
28 +# priority: 60000
29 +
30 +# retries sets the number of retries to be made in case of failures.
31 +# If unset, the default for python.d.plugin is used.
32 +# Attempts to restore the service are made once every update_every
33 +# and only if the module has collected values in the past.
34 +# retries: 60
35 +
36 +# autodetection_retry sets the job re-check interval in seconds.
37 +# The job is not deleted if check fails.
38 +# Attempts to start the job are made once every autodetection_retry.
39 +# This feature is disabled by default.
40 +# autodetection_retry: 0
41 +
42 +# ----------------------------------------------------------------------
43 +# JOBS (data collection sources)
44 +#
45 +# The default JOBS share the same *name*. JOBS with the same name
46 +# are mutually exclusive. Only one of them will be allowed running at
47 +# any time. This allows autodetection to try several alternatives and
48 +# pick the one that works.
49 +#
50 +# Any number of jobs is supported.
51 +#
52 +# All python.d.plugin JOBS (for all its modules) support a set of
53 +# predefined parameters. These are:
54 +#
55 +# job_name:
56 +# name: myname # the JOB's name as it will appear at the
57 +# # dashboard (by default is the job_name)
58 +# # JOBs sharing a name are mutually exclusive
59 +# update_every: 1 # the JOB's data collection frequency
60 +# priority: 60000 # the JOB's order on the dashboard
61 +# retries: 60 # the JOB's number of restoration attempts
62 +# autodetection_retry: 0 # the JOB's re-check interval in seconds
63 +#
64 +# Additionally to the above, example also supports the following:
65 +#
66 +# poll_seconds: SECONDS # default is 1. Sets the frequency of seconds the nvidia-smi tool is polled.
67 +#
68 +# ----------------------------------------------------------------------
collectors/python.d.plugin/python.d.conf
+1
@@ -67,6 +67,7 @@ logind: no
67 # mysql: yes
68 # nginx: yes
69 # nginx_plus: yes
70 +# nvidia_smi: yes
71
72 # nginx_log has been replaced by web_log
73 nginx_log: no