Mark Bergsma has uploaded a new change for review.

  https://gerrit.wikimedia.org/r/53569


Change subject: Initial attempt at a new Ganglia module
......................................................................

Initial attempt at a new Ganglia module

Untested. Moves to a model with central aggregators, and
no multicast.

Change-Id: I0a00cc0ea51577b98332eac422a7ed1a1626ade6
---
A modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf
A modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf
A modules/ganglia/files/upstart/ganglia-monitor.conf
A modules/ganglia/manifests/configuration.pp
A modules/ganglia/manifests/monitor/aggregator/init.pp
A modules/ganglia/manifests/monitor/aggregator/instance.pp
A modules/ganglia/manifests/monitor/config.pp
A modules/ganglia/manifests/monitor/init.pp
A modules/ganglia/manifests/monitor/packages.pp
A modules/ganglia/manifests/monitor/service.pp
A modules/ganglia/templates/gmond.conf.erb
11 files changed, 604 insertions(+), 0 deletions(-)


  git pull ssh://gerrit.wikimedia.org:29418/operations/puppet 
refs/changes/69/53569/1

diff --git 
a/modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf 
b/modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf
new file mode 100644
index 0000000..21e9e6c
--- /dev/null
+++ b/modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf
@@ -0,0 +1,14 @@
+# files/ganglia-monitor-aggregator-instance.conf
+
+description "Ganglia Monitor aggregator instance"
+author "Mark Bergsma <[email protected]>"
+
+stop on stopping ganglia-monitor-aggregator
+
+instance $ID
+
+expect daemon
+respawn
+respawn limit 10 5
+
+exec /usr/sbin/gmond -c /etc/ganglia/aggregators/$ID.conf -p 
/var/run/gmond-$ID.pid
\ No newline at end of file
diff --git a/modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf 
b/modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf
new file mode 100644
index 0000000..84a554a
--- /dev/null
+++ b/modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf
@@ -0,0 +1,16 @@
+# files/ganglia-monitor-aggregator.conf
+
+description "Ganglia Monitor aggregator"
+author "Mark Bergsma <[email protected]>"
+
+start on runlevel [2345]
+stop on runlevel [!2345]
+
+task
+
+script
+       for gmonid in $(ls /etc/ganglia/aggregators/*.conf)
+       do
+               start ganglia-monitor-aggregator-instance ID=${gmonid%.conf}
+       done
+end script
diff --git a/modules/ganglia/files/upstart/ganglia-monitor.conf 
b/modules/ganglia/files/upstart/ganglia-monitor.conf
new file mode 100644
index 0000000..5c36c93
--- /dev/null
+++ b/modules/ganglia/files/upstart/ganglia-monitor.conf
@@ -0,0 +1,13 @@
+# files/ganglia-monitor.conf
+
+description "Ganglia Monitor daemon"
+author "Mark Bergsma <[email protected]>"
+
+start on runlevel [2345]
+stop on runlevel [!2345]
+
+expect daemon
+respawn
+respawn limit 10 5
+
+exec /usr/sbin/gmond
\ No newline at end of file
diff --git a/modules/ganglia/manifests/configuration.pp 
b/modules/ganglia/manifests/configuration.pp
new file mode 100644
index 0000000..dc4e58a
--- /dev/null
+++ b/modules/ganglia/manifests/configuration.pp
@@ -0,0 +1,115 @@
+# modules/ganglia/manifests/configuration.pp
+
+class ganglia::configuration {
+       # NOTE: Do *not* add new clusters *per site* anymore,
+       # the site name will automatically be appended now,
+       # and a different IP prefix will be used.
+       $clusters = {
+               "decommissioned" => {
+                       "name"          => "Decommissioned servers",
+                       "id"    => 1 },
+               "lvs" => {
+                       "name"          => "LVS loadbalancers",
+                       "id"    => 2 },
+               "search"        =>      {
+                       "name"          => "Search",
+                       "id"    => 4 },
+               "mysql"         =>      {
+                       "name"          => "MySQL",
+                       "id"    => 5 },
+               "squids_upload" =>      {
+                       "name"          => "Upload squids",
+                       "id"    => 6 },
+               "squids_text"   =>      {
+                       "name"          => "Text squids",
+                       "id"    => 7 },
+               "misc"          =>      {
+                       "name"          => "Miscellaneous",
+                       "id"    => 8 },
+               "appserver"     =>      {
+                       "name"          => "Application servers",
+                       "id"    => 11   },
+               "imagescaler"   =>      {
+                       "name"          => "Image scalers",
+                       "id"    => 12 },
+               "api_appserver" =>      {
+                       "name"          => "API application servers",
+                       "id"    => 13 },
+               "pdf"           =>      {
+                       "name"          => "PDF servers",
+                       "id"    => 15 },
+               "cache_text"    => {
+                       "name"          => "Text caches",
+                       "id"    => 20 },
+               "cache_bits"    => {
+                       "name"          => "Bits caches",
+                       "id"    => 21 },
+               "cache_upload"  => {
+                       "name"          => "Upload caches",
+                       "id"    => 22 },
+               "payments"      => {
+                       "name"          => "Fundraiser payments",
+                       "id"    => 23 },
+               "bits_appserver"        => {
+                       "name"          => "Bits application servers",
+                       "id"    => 24 },
+               "squids_api"    => {
+                       "name"          => "API squids",
+                       "id"    => 25 },
+               "ssl"           => {
+                       "name"          => "SSL cluster",
+                       "id"    => 26 },
+               "swift" => {
+                       "name"          => "Swift",
+                       "id"    => 27 },
+               "cache_mobile"  => {
+                       "name"          => "Mobile caches",
+                       "id"    => 28 },
+               "virt"  => {
+                       "name"          => "Virtualization cluster",
+                       "id"    => 29 },
+               "gluster"       => {
+                       "name"          => "Glusterfs cluster",
+                       "id"    => 30 },
+               "jobrunner"     =>      {
+                       "name"          => "Jobrunners",
+                       "id"    => 31 },
+               "analytics"             => {
+                       "name"          => "Analytics cluster",
+                       "id"    => 32 },
+               "memcached"             => {
+                       "name"          => "Memcached",
+                       "id"    => 33 },
+               "videoscaler"   => {
+                       "name"          => "Video scalers",
+                       "id"    => 34 },
+               "fundraising"   => {
+                       "name"          => "Fundraising",
+                       "id"    => 35 },
+               "ceph"                  => {
+                       "name"          => "Ceph",
+                       "id"    => 36 },
+               "parsoid"               => {
+                       "name"          => "Parsoid",
+                       "id"    => 37 },
+               "parsoidcache"  => {
+                       "name"          => "Parsoid Varnish",
+                       "id"    => 38 },
+       }
+       # NOTE: Do *not* add new clusters *per site* anymore,
+       # the site name will automatically be appended now,
+       # and a different IP prefix will be used.
+
+       case $::realm {
+               'production': {
+                       $url = "http://ganglia.wikimedia.org";
+                       $gmetad_hosts = [ "208.80.152.15" ]
+                       $base_port = 8649
+               },
+               'labs': {
+                       $url = "http://ganglia.wmflabs.org";
+                       $gmetad_hosts = [ "10.4.0.79"]
+                       $base_port = 8649
+               }
+       }
+}
\ No newline at end of file
diff --git a/modules/ganglia/manifests/monitor/aggregator/init.pp 
b/modules/ganglia/manifests/monitor/aggregator/init.pp
new file mode 100644
index 0000000..e1e533b
--- /dev/null
+++ b/modules/ganglia/manifests/monitor/aggregator/init.pp
@@ -0,0 +1,26 @@
+class ganglia::monitor::aggregator {
+       require ganglia::monitor::packages
+
+       file {
+               "/etc/ganglia/aggregators":
+                       ensure => directory,
+                       mode => 0555;
+               "/etc/init/ganglia-monitor-aggregator.conf":
+                       source => 
"puppet:///modules/ganglia/upstart/ganglia-monitor-aggregator.conf",
+                       mode => 0444;
+               "/etc/init/ganglia-monitor-aggregator-instance.conf":
+                       source => 
"puppet:///modules/ganglia/upstart/ganglia-monitor-aggregator-instance.conf",
+                       mode => 0444;
+       }
+
+       upstart_job { "ganglia-monitor-aggregator-instance": }
+
+       # Instantiate aggregators for all clusters
+       instance{ keys($ganglia::configuration::clusters): }
+
+       service { "ganglia-monitor-aggregator":
+               provider => upstart,
+               name => "ganglia-monitor-aggregator",
+               ensure => running
+       }
+}
diff --git a/modules/ganglia/manifests/monitor/aggregator/instance.pp 
b/modules/ganglia/manifests/monitor/aggregator/instance.pp
new file mode 100644
index 0000000..176d040
--- /dev/null
+++ b/modules/ganglia/manifests/monitor/aggregator/instance.pp
@@ -0,0 +1,20 @@
+define ganglia::monitor::aggregator::instance() {
+       Class[ganglia::monitor::aggregator] -> 
Ganglia::Monitor::Aggregator::Instance[$title]
+       Ganglia::Monitor::Aggregator::Instance[$title] -> 
Service[ganglia-monitor-aggregator]
+
+       $aggregator = true
+
+       # TODO: support multiple $site
+       $cluster = $title
+       $id = $ganglia::configuration::clusters[$cluster]['id']
+       $gmond_port = $::realm ? { 
+               production => $ganglia::configuration::base_port + $id,
+               labs => $::project_gid
+       }
+
+       file { "/etc/ganglia/aggregators/${id}.conf":
+               mode => 0444,
+               content => template("ganglia/gmond.conf.erb"),
+               notify => Service[$title]
+       }
+}
diff --git a/modules/ganglia/manifests/monitor/config.pp 
b/modules/ganglia/manifests/monitor/config.pp
new file mode 100644
index 0000000..86b7005
--- /dev/null
+++ b/modules/ganglia/manifests/monitor/config.pp
@@ -0,0 +1,16 @@
+class ganglia::monitor::config($cluster) {
+       require ganglia::monitor::packages
+
+       $aggregator = false
+       $id = $ganglia::configuration::clusters[$cluster]['id']
+       $gmond_port = $::realm ? { 
+               production => $ganglia::configuration::base_port + $id,
+               labs => $::project_gid
+       }
+
+       file { "/etc/ganglia/gmond.conf":
+               mode => 0444,
+               content => template("ganglia/gmond.conf.erb"),
+               notify => Service["ganglia-monitor"]
+       }
+}
diff --git a/modules/ganglia/manifests/monitor/init.pp 
b/modules/ganglia/manifests/monitor/init.pp
new file mode 100644
index 0000000..c2f7ad6
--- /dev/null
+++ b/modules/ganglia/manifests/monitor/init.pp
@@ -0,0 +1,5 @@
+class ganglia::monitor($cluster) {
+       include packages, service
+
+       class { "ganglia::monitor::config": cluster => $cluster }
+}
\ No newline at end of file
diff --git a/modules/ganglia/manifests/monitor/packages.pp 
b/modules/ganglia/manifests/monitor/packages.pp
new file mode 100644
index 0000000..75d4610
--- /dev/null
+++ b/modules/ganglia/manifests/monitor/packages.pp
@@ -0,0 +1,10 @@
+class ganglia::monitor::packages {
+       package { "ganglia-monitor": ensure => latest }
+       
+       file { "/etc/init/ganglia-monitor.conf":
+               source => 
"puppet:///modules/ganglia/upstart/ganglia-monitor.conf",
+               mode => 0444
+       }
+       
+       upstart_job { "ganglia-monitor": }
+}
diff --git a/modules/ganglia/manifests/monitor/service.pp 
b/modules/ganglia/manifests/monitor/service.pp
new file mode 100644
index 0000000..6171fdf
--- /dev/null
+++ b/modules/ganglia/manifests/monitor/service.pp
@@ -0,0 +1,8 @@
+class ganglia::monitor::service() {
+       Class[ganglia::monitor::config] -> Class[ganglia::monitor::service]
+
+       service { "ganglia-monitor":
+               ensure => running,
+               provider => upstart
+       }
+}
diff --git a/modules/ganglia/templates/gmond.conf.erb 
b/modules/ganglia/templates/gmond.conf.erb
new file mode 100644
index 0000000..f10771f
--- /dev/null
+++ b/modules/ganglia/templates/gmond.conf.erb
@@ -0,0 +1,361 @@
+# This file is managed by Puppet!
+
+/* This configuration is as close to 2.5.x default behavior as possible
+   The values closely match ./gmond/metric.h definitions in 2.5.x */
+globals {
+  daemonize = yes
+  setuid = yes
+  user = ganglia
+  debug_level = 0
+  max_udp_msg_len = 1472
+  mute = <%= aggregator ? "yes": "no" %>
+  deaf = <%= aggregator ? "no": "yes" %>
+  host_dmax = 86400 /*secs */
+  cleanup_threshold = 300 /*secs */
+  gexec = no
+  send_metadata_interval = 60
+}
+
+/* If a cluster attribute is specified, then all gmond hosts are wrapped inside
+ * of a <CLUSTER> tag.  If you do not specify a cluster tag, then all <HOSTS> 
will
+ * NOT be wrapped inside of a <CLUSTER> tag. */
+cluster {
+  name = "<%= cluster %>"
+  owner = "Wikimedia Foundation"
+  latlong = "unspecified"
+  url = "<%= scope.lookupvar("ganglia::configuration::url") %>"
+}
+
+/* The host section describes attributes of the host, like the location */
+host {
+  location = "<%= scope.lookupvar("::site") %>"
+}
+
+<% if not aggregator -%>
+<% scope.lookupvar("ganglia::configuration::gmetad_hosts").each do |host| -%>
+udp_send_channel {
+  host = <%= host %>
+  port = <%= gmond_port %>
+}
+<% end -%>
+<% else -%>
+/* You can specify as many udp_recv_channels as you like as well. */
+udp_recv_channel {
+  port = <%= gmond_port %>
+  acl {
+    default = "deny"
+<% scope.lookupvar("::network::all_networks").each do |prefix|
+ip, mask = prefix.split('/')
+-%>
+    acess {
+      ip = <%= ip %>
+      mask = <%= mask %>
+      action = "allow"
+    }
+<% end -%>
+  }
+}
+
+/* You can specify as many tcp_accept_channels as you like to share
+   an xml description of the state of the cluster */
+tcp_accept_channel {
+  port = <%= gmond_port %>
+  acl {
+    default = "deny"
+<% scope.lookupvar("ganglia::configuration::gmetad_hosts").each do |host|
+-%>
+    access {
+      ip = <%= host %>
+      mask = 32
+      action = "allow"
+    }
+<% end -%>
+  }
+}
+<% end -%>
+
+/* Each metrics module that is referenced by gmond must be specified and
+   loaded. If the module has been statically linked with gmond, it does not
+   require a load path. However all dynamically loadable modules must include
+   a load path. */
+modules {
+  module {
+    name = "core_metrics"
+  }
+  module {
+    name = "cpu_module"
+    path = "/usr/lib/ganglia/modcpu.so"
+  }
+  module {
+    name = "disk_module"
+    path = "/usr/lib/ganglia/moddisk.so"
+  }
+  module {
+    name = "load_module"
+    path = "/usr/lib/ganglia/modload.so"
+  }
+  module {
+    name = "mem_module"
+    path = "/usr/lib/ganglia/modmem.so"
+  }
+  module {
+    name = "net_module"
+    path = "/usr/lib/ganglia/modnet.so"
+  }
+  module {
+    name = "proc_module"
+    path = "/usr/lib/ganglia/modproc.so"
+  }
+  module {
+    name = "sys_module"
+    path = "/usr/lib/ganglia/modsys.so"
+  }
+}
+
+include ('/etc/ganglia/conf.d/*.conf')
+
+
+/* The old internal 2.5.x metric array has been replaced by the following
+   collection_group directives.  What follows is the default behavior for
+   collecting and sending metrics that is as close to 2.5.x behavior as
+   possible. */
+
+/* This collection group will cause a heartbeat (or beacon) to be sent every
+   20 seconds.  In the heartbeat is the GMOND_STARTED data which expresses
+   the age of the running gmond. */
+collection_group {
+  collect_once = yes
+  time_threshold = 20
+  metric {
+    name = "heartbeat"
+  }
+}
+
+/* This collection group will send general info about this host every 1200 
secs.
+   This information doesn't change between reboots and is only collected once. 
*/
+collection_group {
+  collect_once = yes
+  time_threshold = 1200
+  metric {
+    name = "cpu_num"
+    title = "CPU Count"
+  }
+  metric {
+    name = "cpu_speed"
+    title = "CPU Speed"
+  }
+  metric {
+    name = "mem_total"
+    title = "Memory Total"
+  }
+  /* Should this be here? Swap can be added/removed between reboots. */
+  metric {
+    name = "swap_total"
+    title = "Swap Space Total"
+  }
+  metric {
+    name = "boottime"
+    title = "Last Boot Time"
+  }
+  metric {
+    name = "machine_type"
+    title = "Machine Type"
+  }
+  metric {
+    name = "os_name"
+    title = "Operating System"
+  }
+  metric {
+    name = "os_release"
+    title = "Operating System Release"
+  }
+  metric {
+    name = "location"
+    title = "Location"
+  }
+}
+
+/* This collection group will send the status of gexecd for this host every 
300 secs */
+/* Unlike 2.5.x the default behavior is to report gexecd OFF.  */
+collection_group {
+  collect_once = yes
+  time_threshold = 300
+  metric {
+    name = "gexec"
+    title = "Gexec Status"
+  }
+}
+
+/* This collection group will collect the CPU status info every 20 secs.
+   The time threshold is set to 90 seconds.  In honesty, this time_threshold 
could be
+   set significantly higher to reduce unneccessary network chatter. */
+collection_group {
+  collect_every = 20
+  time_threshold = 90
+  /* CPU status */
+  metric {
+    name = "cpu_user"
+    value_threshold = "1.0"
+    title = "CPU User"
+  }
+  metric {
+    name = "cpu_system"
+    value_threshold = "1.0"
+    title = "CPU System"
+  }
+  metric {
+    name = "cpu_idle"
+    value_threshold = "5.0"
+    title = "CPU Idle"
+  }
+  metric {
+    name = "cpu_nice"
+    value_threshold = "1.0"
+    title = "CPU Nice"
+  }
+  metric {
+    name = "cpu_aidle"
+    value_threshold = "5.0"
+    title = "CPU aidle"
+  }
+  metric {
+    name = "cpu_wio"
+    value_threshold = "1.0"
+    title = "CPU wio"
+  }
+  /* The next two metrics are optional if you want more detail...
+     ... since they are accounted for in cpu_system.
+  metric {
+    name = "cpu_intr"
+    value_threshold = "1.0"
+    title = "CPU intr"
+  }
+  metric {
+    name = "cpu_sintr"
+    value_threshold = "1.0"
+    title = "CPU sintr"
+  }
+  */
+}
+
+collection_group {
+  collect_every = 20
+  time_threshold = 90
+  /* Load Averages */
+  metric {
+    name = "load_one"
+    value_threshold = "1.0"
+    title = "One Minute Load Average"
+  }
+  metric {
+    name = "load_five"
+    value_threshold = "1.0"
+    title = "Five Minute Load Average"
+  }
+  metric {
+    name = "load_fifteen"
+    value_threshold = "1.0"
+    title = "Fifteen Minute Load Average"
+  }
+}
+
+/* This group collects the number of running and total processes */
+collection_group {
+  collect_every = 80
+  time_threshold = 950
+  metric {
+    name = "proc_run"
+    value_threshold = "1.0"
+    title = "Total Running Processes"
+  }
+  metric {
+    name = "proc_total"
+    value_threshold = "1.0"
+    title = "Total Processes"
+  }
+}
+
+/* This collection group grabs the volatile memory metrics every 40 secs and
+   sends them at least every 180 secs.  This time_threshold can be increased
+   significantly to reduce unneeded network traffic. */
+collection_group {
+  collect_every = 40
+  time_threshold = 180
+  metric {
+    name = "mem_free"
+    value_threshold = "1024.0"
+    title = "Free Memory"
+  }
+  metric {
+    name = "mem_shared"
+    value_threshold = "1024.0"
+    title = "Shared Memory"
+  }
+  metric {
+    name = "mem_buffers"
+    value_threshold = "1024.0"
+    title = "Memory Buffers"
+  }
+  metric {
+    name = "mem_cached"
+    value_threshold = "1024.0"
+    title = "Cached Memory"
+  }
+  metric {
+    name = "swap_free"
+    value_threshold = "1024.0"
+    title = "Free Swap Space"
+  }
+}
+
+collection_group {
+  collect_every = 40
+  time_threshold = 300
+  metric {
+    name = "bytes_out"
+    value_threshold = 4096
+    title = "Bytes Sent"
+  }
+  metric {
+    name = "bytes_in"
+    value_threshold = 4096
+    title = "Bytes Received"
+  }
+  metric {
+    name = "pkts_in"
+    value_threshold = 256
+    title = "Packets Received"
+  }
+  metric {
+    name = "pkts_out"
+    value_threshold = 256
+    title = "Packets Sent"
+  }
+}
+
+/* Different than 2.5.x default since the old config made no sense */
+collection_group {
+  collect_every = 1800
+  time_threshold = 3600
+  metric {
+    name = "disk_total"
+    value_threshold = 1.0
+    title = "Total Disk Space"
+  }
+}
+
+collection_group {
+  collect_every = 40
+  time_threshold = 180
+  metric {
+    name = "disk_free"
+    value_threshold = 1.0
+    title = "Disk Space Available"
+  }
+  metric {
+    name = "part_max_used"
+    value_threshold = 1.0
+    title = "Maximum Disk Space Used"
+  }
+}
+

-- 
To view, visit https://gerrit.wikimedia.org/r/53569
To unsubscribe, visit https://gerrit.wikimedia.org/r/settings

Gerrit-MessageType: newchange
Gerrit-Change-Id: I0a00cc0ea51577b98332eac422a7ed1a1626ade6
Gerrit-PatchSet: 1
Gerrit-Project: operations/puppet
Gerrit-Branch: production
Gerrit-Owner: Mark Bergsma <[email protected]>

_______________________________________________
MediaWiki-commits mailing list
[email protected]
https://lists.wikimedia.org/mailman/listinfo/mediawiki-commits

Reply via email to