Mark Bergsma has uploaded a new change for review. https://gerrit.wikimedia.org/r/53569
Change subject: Initial attempt at a new Ganglia module ...................................................................... Initial attempt at a new Ganglia module Untested. Moves to a model with central aggregators, and no multicast. Change-Id: I0a00cc0ea51577b98332eac422a7ed1a1626ade6 --- A modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf A modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf A modules/ganglia/files/upstart/ganglia-monitor.conf A modules/ganglia/manifests/configuration.pp A modules/ganglia/manifests/monitor/aggregator/init.pp A modules/ganglia/manifests/monitor/aggregator/instance.pp A modules/ganglia/manifests/monitor/config.pp A modules/ganglia/manifests/monitor/init.pp A modules/ganglia/manifests/monitor/packages.pp A modules/ganglia/manifests/monitor/service.pp A modules/ganglia/templates/gmond.conf.erb 11 files changed, 604 insertions(+), 0 deletions(-) git pull ssh://gerrit.wikimedia.org:29418/operations/puppet refs/changes/69/53569/1 diff --git a/modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf b/modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf new file mode 100644 index 0000000..21e9e6c --- /dev/null +++ b/modules/ganglia/files/upstart/ganglia-monitor-aggregator-instance.conf @@ -0,0 +1,14 @@ +# files/ganglia-monitor-aggregator-instance.conf + +description "Ganglia Monitor aggregator instance" +author "Mark Bergsma <[email protected]>" + +stop on stopping ganglia-monitor-aggregator + +instance $ID + +expect daemon +respawn +respawn limit 10 5 + +exec /usr/sbin/gmond -c /etc/ganglia/aggregators/$ID.conf -p /var/run/gmond-$ID.pid \ No newline at end of file diff --git a/modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf b/modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf new file mode 100644 index 0000000..84a554a --- /dev/null +++ b/modules/ganglia/files/upstart/ganglia-monitor-aggregator.conf @@ -0,0 +1,16 @@ +# files/ganglia-monitor-aggregator.conf + +description "Ganglia Monitor aggregator" +author "Mark Bergsma <[email protected]>" + +start on runlevel [2345] +stop on runlevel [!2345] + +task + +script + for gmonid in $(ls /etc/ganglia/aggregators/*.conf) + do + start ganglia-monitor-aggregator-instance ID=${gmonid%.conf} + done +end script diff --git a/modules/ganglia/files/upstart/ganglia-monitor.conf b/modules/ganglia/files/upstart/ganglia-monitor.conf new file mode 100644 index 0000000..5c36c93 --- /dev/null +++ b/modules/ganglia/files/upstart/ganglia-monitor.conf @@ -0,0 +1,13 @@ +# files/ganglia-monitor.conf + +description "Ganglia Monitor daemon" +author "Mark Bergsma <[email protected]>" + +start on runlevel [2345] +stop on runlevel [!2345] + +expect daemon +respawn +respawn limit 10 5 + +exec /usr/sbin/gmond \ No newline at end of file diff --git a/modules/ganglia/manifests/configuration.pp b/modules/ganglia/manifests/configuration.pp new file mode 100644 index 0000000..dc4e58a --- /dev/null +++ b/modules/ganglia/manifests/configuration.pp @@ -0,0 +1,115 @@ +# modules/ganglia/manifests/configuration.pp + +class ganglia::configuration { + # NOTE: Do *not* add new clusters *per site* anymore, + # the site name will automatically be appended now, + # and a different IP prefix will be used. + $clusters = { + "decommissioned" => { + "name" => "Decommissioned servers", + "id" => 1 }, + "lvs" => { + "name" => "LVS loadbalancers", + "id" => 2 }, + "search" => { + "name" => "Search", + "id" => 4 }, + "mysql" => { + "name" => "MySQL", + "id" => 5 }, + "squids_upload" => { + "name" => "Upload squids", + "id" => 6 }, + "squids_text" => { + "name" => "Text squids", + "id" => 7 }, + "misc" => { + "name" => "Miscellaneous", + "id" => 8 }, + "appserver" => { + "name" => "Application servers", + "id" => 11 }, + "imagescaler" => { + "name" => "Image scalers", + "id" => 12 }, + "api_appserver" => { + "name" => "API application servers", + "id" => 13 }, + "pdf" => { + "name" => "PDF servers", + "id" => 15 }, + "cache_text" => { + "name" => "Text caches", + "id" => 20 }, + "cache_bits" => { + "name" => "Bits caches", + "id" => 21 }, + "cache_upload" => { + "name" => "Upload caches", + "id" => 22 }, + "payments" => { + "name" => "Fundraiser payments", + "id" => 23 }, + "bits_appserver" => { + "name" => "Bits application servers", + "id" => 24 }, + "squids_api" => { + "name" => "API squids", + "id" => 25 }, + "ssl" => { + "name" => "SSL cluster", + "id" => 26 }, + "swift" => { + "name" => "Swift", + "id" => 27 }, + "cache_mobile" => { + "name" => "Mobile caches", + "id" => 28 }, + "virt" => { + "name" => "Virtualization cluster", + "id" => 29 }, + "gluster" => { + "name" => "Glusterfs cluster", + "id" => 30 }, + "jobrunner" => { + "name" => "Jobrunners", + "id" => 31 }, + "analytics" => { + "name" => "Analytics cluster", + "id" => 32 }, + "memcached" => { + "name" => "Memcached", + "id" => 33 }, + "videoscaler" => { + "name" => "Video scalers", + "id" => 34 }, + "fundraising" => { + "name" => "Fundraising", + "id" => 35 }, + "ceph" => { + "name" => "Ceph", + "id" => 36 }, + "parsoid" => { + "name" => "Parsoid", + "id" => 37 }, + "parsoidcache" => { + "name" => "Parsoid Varnish", + "id" => 38 }, + } + # NOTE: Do *not* add new clusters *per site* anymore, + # the site name will automatically be appended now, + # and a different IP prefix will be used. + + case $::realm { + 'production': { + $url = "http://ganglia.wikimedia.org" + $gmetad_hosts = [ "208.80.152.15" ] + $base_port = 8649 + }, + 'labs': { + $url = "http://ganglia.wmflabs.org" + $gmetad_hosts = [ "10.4.0.79"] + $base_port = 8649 + } + } +} \ No newline at end of file diff --git a/modules/ganglia/manifests/monitor/aggregator/init.pp b/modules/ganglia/manifests/monitor/aggregator/init.pp new file mode 100644 index 0000000..e1e533b --- /dev/null +++ b/modules/ganglia/manifests/monitor/aggregator/init.pp @@ -0,0 +1,26 @@ +class ganglia::monitor::aggregator { + require ganglia::monitor::packages + + file { + "/etc/ganglia/aggregators": + ensure => directory, + mode => 0555; + "/etc/init/ganglia-monitor-aggregator.conf": + source => "puppet:///modules/ganglia/upstart/ganglia-monitor-aggregator.conf", + mode => 0444; + "/etc/init/ganglia-monitor-aggregator-instance.conf": + source => "puppet:///modules/ganglia/upstart/ganglia-monitor-aggregator-instance.conf", + mode => 0444; + } + + upstart_job { "ganglia-monitor-aggregator-instance": } + + # Instantiate aggregators for all clusters + instance{ keys($ganglia::configuration::clusters): } + + service { "ganglia-monitor-aggregator": + provider => upstart, + name => "ganglia-monitor-aggregator", + ensure => running + } +} diff --git a/modules/ganglia/manifests/monitor/aggregator/instance.pp b/modules/ganglia/manifests/monitor/aggregator/instance.pp new file mode 100644 index 0000000..176d040 --- /dev/null +++ b/modules/ganglia/manifests/monitor/aggregator/instance.pp @@ -0,0 +1,20 @@ +define ganglia::monitor::aggregator::instance() { + Class[ganglia::monitor::aggregator] -> Ganglia::Monitor::Aggregator::Instance[$title] + Ganglia::Monitor::Aggregator::Instance[$title] -> Service[ganglia-monitor-aggregator] + + $aggregator = true + + # TODO: support multiple $site + $cluster = $title + $id = $ganglia::configuration::clusters[$cluster]['id'] + $gmond_port = $::realm ? { + production => $ganglia::configuration::base_port + $id, + labs => $::project_gid + } + + file { "/etc/ganglia/aggregators/${id}.conf": + mode => 0444, + content => template("ganglia/gmond.conf.erb"), + notify => Service[$title] + } +} diff --git a/modules/ganglia/manifests/monitor/config.pp b/modules/ganglia/manifests/monitor/config.pp new file mode 100644 index 0000000..86b7005 --- /dev/null +++ b/modules/ganglia/manifests/monitor/config.pp @@ -0,0 +1,16 @@ +class ganglia::monitor::config($cluster) { + require ganglia::monitor::packages + + $aggregator = false + $id = $ganglia::configuration::clusters[$cluster]['id'] + $gmond_port = $::realm ? { + production => $ganglia::configuration::base_port + $id, + labs => $::project_gid + } + + file { "/etc/ganglia/gmond.conf": + mode => 0444, + content => template("ganglia/gmond.conf.erb"), + notify => Service["ganglia-monitor"] + } +} diff --git a/modules/ganglia/manifests/monitor/init.pp b/modules/ganglia/manifests/monitor/init.pp new file mode 100644 index 0000000..c2f7ad6 --- /dev/null +++ b/modules/ganglia/manifests/monitor/init.pp @@ -0,0 +1,5 @@ +class ganglia::monitor($cluster) { + include packages, service + + class { "ganglia::monitor::config": cluster => $cluster } +} \ No newline at end of file diff --git a/modules/ganglia/manifests/monitor/packages.pp b/modules/ganglia/manifests/monitor/packages.pp new file mode 100644 index 0000000..75d4610 --- /dev/null +++ b/modules/ganglia/manifests/monitor/packages.pp @@ -0,0 +1,10 @@ +class ganglia::monitor::packages { + package { "ganglia-monitor": ensure => latest } + + file { "/etc/init/ganglia-monitor.conf": + source => "puppet:///modules/ganglia/upstart/ganglia-monitor.conf", + mode => 0444 + } + + upstart_job { "ganglia-monitor": } +} diff --git a/modules/ganglia/manifests/monitor/service.pp b/modules/ganglia/manifests/monitor/service.pp new file mode 100644 index 0000000..6171fdf --- /dev/null +++ b/modules/ganglia/manifests/monitor/service.pp @@ -0,0 +1,8 @@ +class ganglia::monitor::service() { + Class[ganglia::monitor::config] -> Class[ganglia::monitor::service] + + service { "ganglia-monitor": + ensure => running, + provider => upstart + } +} diff --git a/modules/ganglia/templates/gmond.conf.erb b/modules/ganglia/templates/gmond.conf.erb new file mode 100644 index 0000000..f10771f --- /dev/null +++ b/modules/ganglia/templates/gmond.conf.erb @@ -0,0 +1,361 @@ +# This file is managed by Puppet! + +/* This configuration is as close to 2.5.x default behavior as possible + The values closely match ./gmond/metric.h definitions in 2.5.x */ +globals { + daemonize = yes + setuid = yes + user = ganglia + debug_level = 0 + max_udp_msg_len = 1472 + mute = <%= aggregator ? "yes": "no" %> + deaf = <%= aggregator ? "no": "yes" %> + host_dmax = 86400 /*secs */ + cleanup_threshold = 300 /*secs */ + gexec = no + send_metadata_interval = 60 +} + +/* If a cluster attribute is specified, then all gmond hosts are wrapped inside + * of a <CLUSTER> tag. If you do not specify a cluster tag, then all <HOSTS> will + * NOT be wrapped inside of a <CLUSTER> tag. */ +cluster { + name = "<%= cluster %>" + owner = "Wikimedia Foundation" + latlong = "unspecified" + url = "<%= scope.lookupvar("ganglia::configuration::url") %>" +} + +/* The host section describes attributes of the host, like the location */ +host { + location = "<%= scope.lookupvar("::site") %>" +} + +<% if not aggregator -%> +<% scope.lookupvar("ganglia::configuration::gmetad_hosts").each do |host| -%> +udp_send_channel { + host = <%= host %> + port = <%= gmond_port %> +} +<% end -%> +<% else -%> +/* You can specify as many udp_recv_channels as you like as well. */ +udp_recv_channel { + port = <%= gmond_port %> + acl { + default = "deny" +<% scope.lookupvar("::network::all_networks").each do |prefix| +ip, mask = prefix.split('/') +-%> + acess { + ip = <%= ip %> + mask = <%= mask %> + action = "allow" + } +<% end -%> + } +} + +/* You can specify as many tcp_accept_channels as you like to share + an xml description of the state of the cluster */ +tcp_accept_channel { + port = <%= gmond_port %> + acl { + default = "deny" +<% scope.lookupvar("ganglia::configuration::gmetad_hosts").each do |host| +-%> + access { + ip = <%= host %> + mask = 32 + action = "allow" + } +<% end -%> + } +} +<% end -%> + +/* Each metrics module that is referenced by gmond must be specified and + loaded. If the module has been statically linked with gmond, it does not + require a load path. However all dynamically loadable modules must include + a load path. */ +modules { + module { + name = "core_metrics" + } + module { + name = "cpu_module" + path = "/usr/lib/ganglia/modcpu.so" + } + module { + name = "disk_module" + path = "/usr/lib/ganglia/moddisk.so" + } + module { + name = "load_module" + path = "/usr/lib/ganglia/modload.so" + } + module { + name = "mem_module" + path = "/usr/lib/ganglia/modmem.so" + } + module { + name = "net_module" + path = "/usr/lib/ganglia/modnet.so" + } + module { + name = "proc_module" + path = "/usr/lib/ganglia/modproc.so" + } + module { + name = "sys_module" + path = "/usr/lib/ganglia/modsys.so" + } +} + +include ('/etc/ganglia/conf.d/*.conf') + + +/* The old internal 2.5.x metric array has been replaced by the following + collection_group directives. What follows is the default behavior for + collecting and sending metrics that is as close to 2.5.x behavior as + possible. */ + +/* This collection group will cause a heartbeat (or beacon) to be sent every + 20 seconds. In the heartbeat is the GMOND_STARTED data which expresses + the age of the running gmond. */ +collection_group { + collect_once = yes + time_threshold = 20 + metric { + name = "heartbeat" + } +} + +/* This collection group will send general info about this host every 1200 secs. + This information doesn't change between reboots and is only collected once. */ +collection_group { + collect_once = yes + time_threshold = 1200 + metric { + name = "cpu_num" + title = "CPU Count" + } + metric { + name = "cpu_speed" + title = "CPU Speed" + } + metric { + name = "mem_total" + title = "Memory Total" + } + /* Should this be here? Swap can be added/removed between reboots. */ + metric { + name = "swap_total" + title = "Swap Space Total" + } + metric { + name = "boottime" + title = "Last Boot Time" + } + metric { + name = "machine_type" + title = "Machine Type" + } + metric { + name = "os_name" + title = "Operating System" + } + metric { + name = "os_release" + title = "Operating System Release" + } + metric { + name = "location" + title = "Location" + } +} + +/* This collection group will send the status of gexecd for this host every 300 secs */ +/* Unlike 2.5.x the default behavior is to report gexecd OFF. */ +collection_group { + collect_once = yes + time_threshold = 300 + metric { + name = "gexec" + title = "Gexec Status" + } +} + +/* This collection group will collect the CPU status info every 20 secs. + The time threshold is set to 90 seconds. In honesty, this time_threshold could be + set significantly higher to reduce unneccessary network chatter. */ +collection_group { + collect_every = 20 + time_threshold = 90 + /* CPU status */ + metric { + name = "cpu_user" + value_threshold = "1.0" + title = "CPU User" + } + metric { + name = "cpu_system" + value_threshold = "1.0" + title = "CPU System" + } + metric { + name = "cpu_idle" + value_threshold = "5.0" + title = "CPU Idle" + } + metric { + name = "cpu_nice" + value_threshold = "1.0" + title = "CPU Nice" + } + metric { + name = "cpu_aidle" + value_threshold = "5.0" + title = "CPU aidle" + } + metric { + name = "cpu_wio" + value_threshold = "1.0" + title = "CPU wio" + } + /* The next two metrics are optional if you want more detail... + ... since they are accounted for in cpu_system. + metric { + name = "cpu_intr" + value_threshold = "1.0" + title = "CPU intr" + } + metric { + name = "cpu_sintr" + value_threshold = "1.0" + title = "CPU sintr" + } + */ +} + +collection_group { + collect_every = 20 + time_threshold = 90 + /* Load Averages */ + metric { + name = "load_one" + value_threshold = "1.0" + title = "One Minute Load Average" + } + metric { + name = "load_five" + value_threshold = "1.0" + title = "Five Minute Load Average" + } + metric { + name = "load_fifteen" + value_threshold = "1.0" + title = "Fifteen Minute Load Average" + } +} + +/* This group collects the number of running and total processes */ +collection_group { + collect_every = 80 + time_threshold = 950 + metric { + name = "proc_run" + value_threshold = "1.0" + title = "Total Running Processes" + } + metric { + name = "proc_total" + value_threshold = "1.0" + title = "Total Processes" + } +} + +/* This collection group grabs the volatile memory metrics every 40 secs and + sends them at least every 180 secs. This time_threshold can be increased + significantly to reduce unneeded network traffic. */ +collection_group { + collect_every = 40 + time_threshold = 180 + metric { + name = "mem_free" + value_threshold = "1024.0" + title = "Free Memory" + } + metric { + name = "mem_shared" + value_threshold = "1024.0" + title = "Shared Memory" + } + metric { + name = "mem_buffers" + value_threshold = "1024.0" + title = "Memory Buffers" + } + metric { + name = "mem_cached" + value_threshold = "1024.0" + title = "Cached Memory" + } + metric { + name = "swap_free" + value_threshold = "1024.0" + title = "Free Swap Space" + } +} + +collection_group { + collect_every = 40 + time_threshold = 300 + metric { + name = "bytes_out" + value_threshold = 4096 + title = "Bytes Sent" + } + metric { + name = "bytes_in" + value_threshold = 4096 + title = "Bytes Received" + } + metric { + name = "pkts_in" + value_threshold = 256 + title = "Packets Received" + } + metric { + name = "pkts_out" + value_threshold = 256 + title = "Packets Sent" + } +} + +/* Different than 2.5.x default since the old config made no sense */ +collection_group { + collect_every = 1800 + time_threshold = 3600 + metric { + name = "disk_total" + value_threshold = 1.0 + title = "Total Disk Space" + } +} + +collection_group { + collect_every = 40 + time_threshold = 180 + metric { + name = "disk_free" + value_threshold = 1.0 + title = "Disk Space Available" + } + metric { + name = "part_max_used" + value_threshold = 1.0 + title = "Maximum Disk Space Used" + } +} + -- To view, visit https://gerrit.wikimedia.org/r/53569 To unsubscribe, visit https://gerrit.wikimedia.org/r/settings Gerrit-MessageType: newchange Gerrit-Change-Id: I0a00cc0ea51577b98332eac422a7ed1a1626ade6 Gerrit-PatchSet: 1 Gerrit-Project: operations/puppet Gerrit-Branch: production Gerrit-Owner: Mark Bergsma <[email protected]> _______________________________________________ MediaWiki-commits mailing list [email protected] https://lists.wikimedia.org/mailman/listinfo/mediawiki-commits
