nvidia_smi: init version added (#4589)
* nvidia_smi: init version added * Update nvidia_smi.chart.py * nvidia_smi: don't use __bool__ for checking xml parse result
Ilya Mashchenko committed
Nov 12, 2018 at 23:47 UTC
e448530a3e712f2b503ddc9c0d6ebb9cd7426dd0
6 files changed
+482
collectors/python.d.plugin/Makefile.am
+1
@@ -74,6 +74,7 @@ include monit/Makefile.inc
74
include mysql/Makefile.inc
75
include nginx/Makefile.inc
76
include nginx_plus/Makefile.inc
77
+include nvidia_smi/Makefile.inc
78
include nsd/Makefile.inc
79
include ntpd/Makefile.inc
80
include ovpn_status_log/Makefile.inc
collectors/python.d.plugin/nvidia_smi/Makefile.inc
new
+12
@@ -0,0 +1,12 @@
1
+# SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+# THIS IS NOT A COMPLETE Makefile
4
+# IT IS INCLUDED BY ITS PARENT'S Makefile.am
5
+# IT IS REQUIRED TO REFERENCE ALL FILES RELATIVE TO THE PARENT
6
+
7
+# install these files
8
+dist_python_DATA += nvidia_smi/nvidia_smi.chart.py
9
+dist_pythonconfig_DATA += nvidia_smi/nvidia_smi.conf
10
+
11
+# do not install these files, but include them in the distribution
12
+dist_noinst_DATA += nvidia_smi/README.md nvidia_smi/Makefile.inc
collectors/python.d.plugin/nvidia_smi/README.md
new
+39
@@ -0,0 +1,39 @@
1
+# nvidia_smi
2
+
3
+This module monitors the `nvidia-smi` cli tool.
4
+
5
+**Requirements and Notes:**
6
+
7
+ * You must have the `nvidia-smi` tool installed and your NVIDIA GPU(s) must support the tool. Mostly the newer high end models used for AI / ML and Crypto or Pro range, read more about [nvidia_smi](https://developer.nvidia.com/nvidia-system-management-interface).
8
+
9
+ * You must enable this plugin as its disabled by default due to minor performance issues.
10
+
11
+ * On some systems when the GPU is idle the `nvidia-smi` tool unloads and there is added latency again when it is next queried. If you are running GPUs under constant workload this isn't likely to be an issue.
12
+
13
+ * Currently the `nvidia-smi` tool is being queried via cli. Updating the plugin to use the nvidia c/c++ API directly should resolve this issue. See discussion here: https://github.com/netdata/netdata/pull/4357
14
+
15
+ * Contributions are welcome.
16
+
17
+ * Make sure `netdata` user can execute `/usr/bin/nvidia-smi` or wherever your binary is.
18
+
19
+ * `poll_seconds` is how often in seconds the tool is polled for as an integer.
20
+
21
+It produces:
22
+
23
+1. Per GPU
24
+ * GPU utilization
25
+ * memory allocation
26
+ * memory utilization
27
+ * fan speed
28
+ * power usage
29
+ * temperature
30
+ * clock speed
31
+ * PCI bandwidth
32
+
33
+### configuration
34
+
35
+Sample:
36
+
37
+```yaml
38
+poll_seconds: 1
39
+```
\ No newline at end of file
collectors/python.d.plugin/nvidia_smi/nvidia_smi.chart.py
new
+361
@@ -0,0 +1,361 @@
1
+# -*- coding: utf-8 -*-
2
+# Description: nvidia-smi netdata python.d module
3
+# Original Author: Steven Noonan (tycho)
4
+# Author: Ilya Mashchenko (l2isbad)
5
+
6
+import subprocess
7
+import threading
8
+import xml.etree.ElementTree as et
9
+
10
+from bases.collection import find_binary
11
+from bases.FrameworkServices.SimpleService import SimpleService
12
+
13
+disabled_by_default = True
14
+
15
+
16
+NVIDIA_SMI = 'nvidia-smi'
17
+
18
+EMPTY_ROW = ''
19
+EMPTY_ROW_LIMIT = 500
20
+POLLER_BREAK_ROW = '</nvidia_smi_log>'
21
+
22
+PCI_BANDWIDTH = 'pci_bandwidth'
23
+FAN_SPEED = 'fan_speed'
24
+GPU_UTIL = 'gpu_utilization'
25
+MEM_UTIL = 'mem_utilization'
26
+ENCODER_UTIL = 'encoder_utilization'
27
+MEM_ALLOCATED = 'mem_allocated'
28
+TEMPERATURE = 'temperature'
29
+CLOCKS = 'clocks'
30
+POWER = 'power'
31
+
32
+ORDER = [
33
+ PCI_BANDWIDTH,
34
+ FAN_SPEED,
35
+ GPU_UTIL,
36
+ MEM_UTIL,
37
+ ENCODER_UTIL,
38
+ MEM_ALLOCATED,
39
+ TEMPERATURE,
40
+ CLOCKS,
41
+ POWER,
42
+]
43
+
44
+
45
+def gpu_charts(gpu):
46
+ fam = gpu.full_name()
47
+
48
+ charts = {
49
+ PCI_BANDWIDTH: {
50
+ 'options': [None, 'PCI Express Bandwidth Utilization', 'KB/s', fam, 'nvidia_smi.pci_bandwidth', 'area'],
51
+ 'lines': [
52
+ ['rx_util', 'rx', 'absolute', 1, 1],
53
+ ['tx_util', 'tx', 'absolute', 1, -1],
54
+ ]
55
+ },
56
+ FAN_SPEED: {
57
+ 'options': [None, 'Fan Speed', '%', fam, 'nvidia_smi.fan_speed', 'line'],
58
+ 'lines': [
59
+ ['fan_speed', 'speed'],
60
+ ]
61
+ },
62
+ GPU_UTIL: {
63
+ 'options': [None, 'GPU Utilization', '%', fam, 'nvidia_smi.gpu_utilization', 'line'],
64
+ 'lines': [
65
+ ['gpu_util', 'utilization'],
66
+ ]
67
+ },
68
+ MEM_UTIL: {
69
+ 'options': [None, 'Memory Bandwidth Utilization', '%', fam, 'nvidia_smi.mem_utilization', 'line'],
70
+ 'lines': [
71
+ ['memory_util', 'utilization'],
72
+ ]
73
+ },
74
+ ENCODER_UTIL: {
75
+ 'options': [None, 'Encoder/Decoder Utilization', '%', fam, 'nvidia_smi.encoder_utilization', 'line'],
76
+ 'lines': [
77
+ ['encoder_util', 'encoder'],
78
+ ['decoder_util', 'decoder'],
79
+ ]
80
+ },
81
+ MEM_ALLOCATED: {
82
+ 'options': [None, 'Memory Allocated', 'MB', fam, 'nvidia_smi.memory_allocated', 'line'],
83
+ 'lines': [
84
+ ['fb_memory_usage', 'used'],
85
+ ]
86
+ },
87
+ TEMPERATURE: {
88
+ 'options': [None, 'Temperature', 'celsius', fam, 'nvidia_smi.temperature', 'line'],
89
+ 'lines': [
90
+ ['gpu_temp', 'temp'],
91
+ ]
92
+ },
93
+ CLOCKS: {
94
+ 'options': [None, 'Clock Frequencies', 'MHz', fam, 'nvidia_smi.clocks', 'line'],
95
+ 'lines': [
96
+ ['graphics_clock', 'graphics'],
97
+ ['video_clock', 'video'],
98
+ ['sm_clock', 'sm'],
99
+ ['mem_clock', 'mem'],
100
+ ]
101
+ },
102
+ POWER: {
103
+ 'options': [None, 'Power Utilization', 'Watts', fam, 'nvidia_smi.power', 'line'],
104
+ 'lines': [
105
+ ['power_draw', 'power', 1, 100],
106
+ ]
107
+ },
108
+ }
109
+
110
+ idx = gpu.num
111
+
112
+ order = ['gpu{0}_{1}'.format(idx, v) for v in ORDER]
113
+ charts = dict(('gpu{0}_{1}'.format(idx, k), v) for k, v in charts.items())
114
+
115
+ for chart in charts.values():
116
+ for line in chart['lines']:
117
+ line[0] = 'gpu{0}_{1}'.format(idx, line[0])
118
+
119
+ return order, charts
120
+
121
+
122
+class NvidiaSMI:
123
+ def __init__(self):
124
+ self.command = find_binary(NVIDIA_SMI)
125
+ self.active_proc = None
126
+
127
+ def run_once(self):
128
+ proc = subprocess.Popen([self.command, '-x', '-q'], stdout=subprocess.PIPE)
129
+ stdout, _ = proc.communicate()
130
+ return stdout
131
+
132
+ def run_loop(self, interval):
133
+ if self.active_proc:
134
+ self.kill()
135
+ proc = subprocess.Popen([self.command, '-x', '-q', '-l', str(interval)], stdout=subprocess.PIPE)
136
+ self.active_proc = proc
137
+ return proc.stdout
138
+
139
+ def kill(self):
140
+ if self.active_proc:
141
+ self.active_proc.kill()
142
+ self.active_proc = None
143
+
144
+
145
+class NvidiaSMIPoller(threading.Thread):
146
+ def __init__(self, poll_interval):
147
+ threading.Thread.__init__(self)
148
+ self.daemon = True
149
+
150
+ self.smi = NvidiaSMI()
151
+ self.interval = poll_interval
152
+
153
+ self.lock = threading.RLock()
154
+ self.last_data = str()
155
+ self.exit = False
156
+ self.empty_rows = 0
157
+ self.rows = list()
158
+
159
+ def has_smi(self):
160
+ return bool(self.smi.command)
161
+
162
+ def run_once(self):
163
+ return self.smi.run_once()
164
+
165
+ def run(self):
166
+ out = self.smi.run_loop(self.interval)
167
+
168
+ for row in out:
169
+ if self.exit or self.empty_rows > EMPTY_ROW_LIMIT:
170
+ break
171
+ self.process_row(row)
172
+ self.smi.kill()
173
+
174
+ def process_row(self, row):
175
+ row = row.decode()
176
+ self.empty_rows += (row == EMPTY_ROW)
177
+ self.rows.append(row)
178
+
179
+ if POLLER_BREAK_ROW in row:
180
+ self.lock.acquire()
181
+ self.last_data = '\n'.join(self.rows)
182
+ self.lock.release()
183
+
184
+ self.rows = list()
185
+ self.empty_rows = 0
186
+
187
+ def is_started(self):
188
+ return self.ident is not None
189
+
190
+ def shutdown(self):
191
+ self.exit = True
192
+
193
+ def data(self):
194
+ self.lock.acquire()
195
+ data = self.last_data
196
+ self.lock.release()
197
+ return data
198
+
199
+
200
+def handle_attr_error(method):
201
+ def on_call(*args, **kwargs):
202
+ try:
203
+ return method(*args, **kwargs)
204
+ except AttributeError:
205
+ return None
206
+ return on_call
207
+
208
+
209
+class GPU:
210
+ def __init__(self, num, root):
211
+ self.num = num
212
+ self.root = root
213
+
214
+ def id(self):
215
+ return self.root.get('id')
216
+
217
+ def name(self):
218
+ return self.root.find('product_name').text
219
+
220
+ def full_name(self):
221
+ return 'gpu{0} {1}'.format(self.num, self.name())
222
+
223
+ @handle_attr_error
224
+ def rx_util(self):
225
+ return self.root.find('pci').find('rx_util').text.split()[0]
226
+
227
+ @handle_attr_error
228
+ def tx_util(self):
229
+ return self.root.find('pci').find('tx_util').text.split()[0]
230
+
231
+ @handle_attr_error
232
+ def fan_speed(self):
233
+ return self.root.find('fan_speed').text.split()[0]
234
+
235
+ @handle_attr_error
236
+ def gpu_util(self):
237
+ return self.root.find('utilization').find('gpu_util').text.split()[0]
238
+
239
+ @handle_attr_error
240
+ def memory_util(self):
241
+ return self.root.find('utilization').find('memory_util').text.split()[0]
242
+
243
+ @handle_attr_error
244
+ def encoder_util(self):
245
+ return self.root.find('utilization').find('encoder_util').text.split()[0]
246
+
247
+ @handle_attr_error
248
+ def decoder_util(self):
249
+ return self.root.find('utilization').find('decoder_util').text.split()[0]
250
+
251
+ @handle_attr_error
252
+ def fb_memory_usage(self):
253
+ return self.root.find('fb_memory_usage').find('used').text.split()[0]
254
+
255
+ @handle_attr_error
256
+ def temperature(self):
257
+ return self.root.find('temperature').find('gpu_temp').text.split()[0]
258
+
259
+ @handle_attr_error
260
+ def graphics_clock(self):
261
+ return self.root.find('clocks').find('graphics_clock').text.split()[0]
262
+
263
+ @handle_attr_error
264
+ def video_clock(self):
265
+ return self.root.find('clocks').find('video_clock').text.split()[0]
266
+
267
+ @handle_attr_error
268
+ def sm_clock(self):
269
+ return self.root.find('clocks').find('sm_clock').text.split()[0]
270
+
271
+ @handle_attr_error
272
+ def mem_clock(self):
273
+ return self.root.find('clocks').find('mem_clock').text.split()[0]
274
+
275
+ @handle_attr_error
276
+ def power_draw(self):
277
+ return float(self.root.find('power_readings').find('power_draw').text.split()[0]) * 100
278
+
279
+ def data(self):
280
+ data = {
281
+ 'rx_util': self.rx_util(),
282
+ 'tx_util': self.tx_util(),
283
+ 'fan_speed': self.fan_speed(),
284
+ 'gpu_util': self.gpu_util(),
285
+ 'memory_util': self.memory_util(),
286
+ 'encoder_util': self.encoder_util(),
287
+ 'decoder_util': self.decoder_util(),
288
+ 'fb_memory_usage': self.fb_memory_usage(),
289
+ 'gpu_temp': self.temperature(),
290
+ 'graphics_clock': self.graphics_clock(),
291
+ 'video_clock': self.video_clock(),
292
+ 'sm_clock': self.sm_clock(),
293
+ 'mem_clock': self.mem_clock(),
294
+ 'power_draw': self.power_draw(),
295
+ }
296
+
297
+ return dict(('gpu{0}_{1}'.format(self.num, k), v) for k, v in data.items() if v is not None)
298
+
299
+
300
+class Service(SimpleService):
301
+ def __init__(self, configuration=None, name=None):
302
+ super(Service, self).__init__(configuration=configuration, name=name)
303
+ self.order = list()
304
+ self.definitions = dict()
305
+
306
+ poll = int(configuration.get('poll_seconds', 1))
307
+ self.poller = NvidiaSMIPoller(poll)
308
+
309
+ def get_data(self):
310
+ if not self.poller.is_alive():
311
+ self.debug('poller is off')
312
+ return None
313
+
314
+ last_data = self.poller.data()
315
+
316
+ parsed = self.parse_xml(last_data)
317
+ if parsed is None:
318
+ return None
319
+
320
+ data = dict()
321
+ for idx, root in enumerate(parsed.findall('gpu')):
322
+ data.update(GPU(idx, root).data())
323
+
324
+ return data or None
325
+
326
+ def check(self):
327
+ if not self.poller.has_smi():
328
+ self.error("couldn't find '{0}' binary".format(NVIDIA_SMI))
329
+ return False
330
+
331
+ raw_data = self.poller.run_once()
332
+ if not raw_data:
333
+ self.error("failed to invoke '{0}' binary".format(NVIDIA_SMI))
334
+ return False
335
+
336
+ parsed = self.parse_xml(raw_data)
337
+ if parsed is None:
338
+ return False
339
+
340
+ gpus = parsed.findall('gpu')
341
+ if not gpus:
342
+ return False
343
+
344
+ self.create_charts(gpus)
345
+ self.poller.start()
346
+
347
+ return True
348
+
349
+ def parse_xml(self, data):
350
+ try:
351
+ return et.fromstring(data)
352
+ except et.ParseError as error:
353
+ self.error(error)
354
+
355
+ return None
356
+
357
+ def create_charts(self, gpus):
358
+ for idx, root in enumerate(gpus):
359
+ order, charts = gpu_charts(GPU(idx, root))
360
+ self.order.extend(order)
361
+ self.definitions.update(charts)
collectors/python.d.plugin/nvidia_smi/nvidia_smi.conf
new
+68
@@ -0,0 +1,68 @@
1
+# netdata python.d.plugin configuration for nvidia_smi
2
+#
3
+# This file is in YaML format. Generally the format is:
4
+#
5
+# name: value
6
+#
7
+# There are 2 sections:
8
+# - global variables
9
+# - one or more JOBS
10
+#
11
+# JOBS allow you to collect values from multiple sources.
12
+# Each source will have its own set of charts.
13
+#
14
+# JOB parameters have to be indented (using spaces only, example below).
15
+
16
+# ----------------------------------------------------------------------
17
+# Global Variables
18
+# These variables set the defaults for all JOBs, however each JOB
19
+# may define its own, overriding the defaults.
20
+
21
+# update_every sets the default data collection frequency.
22
+# If unset, the python.d.plugin default is used.
23
+# update_every: 1
24
+
25
+# priority controls the order of charts at the netdata dashboard.
26
+# Lower numbers move the charts towards the top of the page.
27
+# If unset, the default for python.d.plugin is used.
28
+# priority: 60000
29
+
30
+# retries sets the number of retries to be made in case of failures.
31
+# If unset, the default for python.d.plugin is used.
32
+# Attempts to restore the service are made once every update_every
33
+# and only if the module has collected values in the past.
34
+# retries: 60
35
+
36
+# autodetection_retry sets the job re-check interval in seconds.
37
+# The job is not deleted if check fails.
38
+# Attempts to start the job are made once every autodetection_retry.
39
+# This feature is disabled by default.
40
+# autodetection_retry: 0
41
+
42
+# ----------------------------------------------------------------------
43
+# JOBS (data collection sources)
44
+#
45
+# The default JOBS share the same *name*. JOBS with the same name
46
+# are mutually exclusive. Only one of them will be allowed running at
47
+# any time. This allows autodetection to try several alternatives and
48
+# pick the one that works.
49
+#
50
+# Any number of jobs is supported.
51
+#
52
+# All python.d.plugin JOBS (for all its modules) support a set of
53
+# predefined parameters. These are:
54
+#
55
+# job_name:
56
+# name: myname # the JOB's name as it will appear at the
57
+# # dashboard (by default is the job_name)
58
+# # JOBs sharing a name are mutually exclusive
59
+# update_every: 1 # the JOB's data collection frequency
60
+# priority: 60000 # the JOB's order on the dashboard
61
+# retries: 60 # the JOB's number of restoration attempts
62
+# autodetection_retry: 0 # the JOB's re-check interval in seconds
63
+#
64
+# Additionally to the above, example also supports the following:
65
+#
66
+# poll_seconds: SECONDS # default is 1. Sets the frequency of seconds the nvidia-smi tool is polled.
67
+#
68
+# ----------------------------------------------------------------------
collectors/python.d.plugin/python.d.conf
+1
@@ -67,6 +67,7 @@ logind: no
67
# mysql: yes
68
# nginx: yes
69
# nginx_plus: yes
70
+# nvidia_smi: yes
71
72
# nginx_log has been replaced by web_log
73
nginx_log: no