NEW: Puppet Server and Puppet DB basic monitoring
- Added universal module based on Puppet status API - There is room for further improvement to increase metric count
Andrey Galkin committed
May 3, 2018 at 01:06 UTC
758467c6aed3198280d5c37263a7ae2c0f0cd5db
2 files changed
+211
conf.d/python.d/puppet.conf
new
+97
@@ -0,0 +1,97 @@
1
+# netdata python.d.plugin configuration for Puppet Server and Puppet DB
2
+#
3
+# This file is in YaML format. Generally the format is:
4
+#
5
+# name: value
6
+#
7
+# There are 2 sections:
8
+# - global variables
9
+# - one or more JOBS
10
+#
11
+# JOBS allow you to collect values from multiple sources.
12
+# Each source will have its own set of charts.
13
+#
14
+# JOB parameters have to be indented (using spaces only, example below).
15
+
16
+# ----------------------------------------------------------------------
17
+# Global Variables
18
+# These variables set the defaults for all JOBs, however each JOB
19
+# may define its own, overriding the defaults.
20
+
21
+# update_every sets the default data collection frequency.
22
+# If unset, the python.d.plugin default is used.
23
+# update_every: 1
24
+
25
+# priority controls the order of charts at the netdata dashboard.
26
+# Lower numbers move the charts towards the top of the page.
27
+# If unset, the default for python.d.plugin is used.
28
+# priority: 60000
29
+
30
+# retries sets the number of retries to be made in case of failures.
31
+# If unset, the default for python.d.plugin is used.
32
+# Attempts to restore the service are made once every update_every
33
+# and only if the module has collected values in the past.
34
+# retries: 60
35
+
36
+# autodetection_retry sets the job re-check interval in seconds.
37
+# The job is not deleted if check fails.
38
+# Attempts to start the job are made once every autodetection_retry.
39
+# This feature is disabled by default.
40
+# autodetection_retry: 0
41
+
42
+# ----------------------------------------------------------------------
43
+# JOBS (data collection sources)
44
+#
45
+# The default JOBS share the same *name*. JOBS with the same name
46
+# are mutually exclusive. Only one of them will be allowed running at
47
+# any time. This allows autodetection to try several alternatives and
48
+# pick the one that works.
49
+#
50
+# Any number of jobs is supported.
51
+#
52
+# All python.d.plugin JOBS (for all its modules) support a set of
53
+# predefined parameters. These are:
54
+#
55
+# job_name:
56
+# name: myname # the JOB's name as it will appear at the
57
+# # dashboard (by default is the job_name)
58
+# # JOBs sharing a name are mutually exclusive
59
+# update_every: 1 # the JOB's data collection frequency
60
+# priority: 60000 # the JOB's order on the dashboard
61
+# retries: 60 # the JOB's number of restoration attempts
62
+# autodetection_retry: 0 # the JOB's re-check interval in seconds
63
+#
64
+# These configuration comes from UrlService base:
65
+# url: # HTTP or HTTPS URL
66
+# tls_verify: False # Control HTTPS server certificate verification
67
+# tls_ca_file: # Optional CA (bundle) file to use
68
+# tls_cert_file: # Optional client certificate file
69
+# tls_key_file: # Optional client key file
70
+#
71
+# ----------------------------------------------------------------------
72
+# AUTO-DETECTION JOBS
73
+# only one of them will run (they have the same name)
74
+# puppet:
75
+# url: 'https://<FQDN>:8140'
76
+#
77
+
78
+#
79
+# Production configuration should look like below.
80
+#
81
+# NOTE: Usually Puppet Server/DB startup time is VERY long. So, there should
82
+# be quite reasonable retry count.
83
+#
84
+# NOTE: secure PuppetDB config may require client certificate.
85
+# Not applies by default though.
86
+# puppetdb:
87
+# url: 'https://fqdn.example.com:8081'
88
+# tls_cert_file: /path/to/client.crt
89
+# tls_key_file: /path/to/client.key
90
+# autodetection_retry: 1
91
+# retries: 3600
92
+#
93
+# puppetserver:
94
+# url: 'https://fqdn.example.com:8140'
95
+# autodetection_retry: 1
96
+# retries: 3600
97
+#
python.d/puppet.chart.py
new
+114
@@ -0,0 +1,114 @@
1
+# -*- coding: utf-8 -*-
2
+# Description: puppet netdata python.d module
3
+# Author: Andrey Galkin <andrey@futoin.org> (andvgal)
4
+#---
5
+# This module should work both with OpenSource and PE versions
6
+# of PuppetServer and PuppetDB.
7
+#
8
+# NOTE: PuppetDB may be configured to require proper TLS
9
+# client certificate for security reasons. Use tls_key_file
10
+# and tls_cert_file options then.
11
+#---
12
+
13
+from bases.FrameworkServices.UrlService import UrlService
14
+from json import loads
15
+import socket
16
+
17
+update_every = 5
18
+priority = 60000
19
+# very long clojure-based service startup time
20
+retries = 180
21
+
22
+ORDER = [
23
+ 'jvm_heap',
24
+ 'jvm_nonheap',
25
+ 'cpu',
26
+ 'fd_open',
27
+]
28
+CHARTS = {
29
+ 'jvm_heap': {
30
+ 'options': [None, "JVM Heap", "MB", "resources", "puppet.jvm", "area"],
31
+ 'lines': [
32
+ ["jvm_heap_max", 'max', "absolute", 1, 1048576, 'hidden'],
33
+ ["jvm_heap_committed", 'committed', "absolute", 1, 1048576],
34
+ ["jvm_heap_used", 'used', "absolute", 1, 1048576],
35
+ ["jvm_heap_init", 'initial', "absolute", 1, 1048576, 'hidden'],
36
+ ]
37
+ },
38
+ 'jvm_nonheap': {
39
+ 'options': [None, "JVM Non-Heap", "MB", "resources", "puppet.jvm", "area"],
40
+ 'lines': [
41
+ ["jvm_nonheap_max", 'max', "absolute", 1, 1048576, 'hidden'],
42
+ ["jvm_nonheap_committed", 'committed', "absolute", 1, 1048576],
43
+ ["jvm_nonheap_used", 'used', "absolute", 1, 1048576],
44
+ ["jvm_nonheap_init", 'initial', "absolute", 1, 1048576, 'hidden'],
45
+ ]
46
+ },
47
+ 'cpu': {
48
+ 'options': [None, "CPU usage", "descriptors", "resources", "puppet.cpu", "stacked"],
49
+ 'lines': [
50
+ ["cpu_time", 'execution', "absolute", 1, 1000],
51
+ ["gc_time", 'GC', "absolute", 1, 1000],
52
+ ]
53
+ },
54
+ 'fd_open': {
55
+ 'options': [None, "File Descriptors", "descriptors", "resources", "puppet.fdopen", "line"],
56
+ 'lines': [
57
+ ["fd_max", 'max', "absolute", 1, 1, 'hidden'],
58
+ ["fd_used", 'used', "absolute"],
59
+ ]
60
+ },
61
+}
62
+
63
+class Service(UrlService):
64
+ def __init__(self, configuration=None, name=None):
65
+ UrlService.__init__(self, configuration=configuration, name=name)
66
+ self.url = 'https://{0}:8140'.format(socket.getfqdn())
67
+ self.order = ORDER
68
+ self.definitions = CHARTS
69
+
70
+ def _get_data(self):
71
+ #---
72
+ # NOTE: there are several ways to retrieve data
73
+ # 1. Only PE versions:
74
+ # https://puppet.com/docs/pe/2018.1/api_status/status_api_metrics_endpoints.html
75
+ # 2. Inidividual Metrics API (JMX):
76
+ # https://puppet.com/docs/pe/2018.1/api_status/metrics_api.html
77
+ # 3. Extended status at debug level:
78
+ # https://puppet.com/docs/pe/2018.1/api_status/status_api_json_endpoints.html
79
+ #
80
+ # For sake of simplicity and efficiency the status one is used..
81
+ #---
82
+
83
+ raw_data = self._get_raw_data(self.url + '/status/v1/services?level=debug')
84
+
85
+ if raw_data is None:
86
+ return None
87
+
88
+ raw_data = loads(raw_data)
89
+ data = {}
90
+
91
+ try:
92
+ try:
93
+ jvm_metrics = raw_data['status-service']['status']['experimental']['jvm-metrics']
94
+ except KeyError:
95
+ jvm_metrics = raw_data['status-service']['status']['jvm-metrics']
96
+
97
+ heap_mem = jvm_metrics['heap-memory']
98
+ non_heap_mem = jvm_metrics['non-heap-memory']
99
+
100
+ for k in ['max', 'committed', 'used', 'init']:
101
+ data['jvm_heap_'+k] = heap_mem[k]
102
+ data['jvm_nonheap_'+k] = non_heap_mem[k]
103
+
104
+ fd_open = jvm_metrics['file-descriptors']
105
+ data['fd_max'] = fd_open['max']
106
+ data['fd_used'] = fd_open['used']
107
+
108
+ data['cpu_time'] = int(jvm_metrics['cpu-usage'] * 1000)
109
+ data['gc_time'] = int(jvm_metrics['gc-cpu-usage'] * 1000)
110
+ except KeyError:
111
+ pass
112
+
113
+
114
+ return data or None