Linux服务器硬件运行状态及故障邮件提醒的监控脚本分享
程序员文章站
2022-11-13 12:21:15
监控硬件运行状况
shell 监控cpu,memory,load average,记录到log,当负载压力时,发电邮通知管理员。
原理:
1.获取cpu,memory...
监控硬件运行状况
shell 监控cpu,memory,load average,记录到log,当负载压力时,发电邮通知管理员。
原理:
1.获取cpu,memory,load average的数值
2.判断数值是否超过自定义的范围,例如(cpu>90%,memory<10%,load average>2)
3.如数值超过范围,发送电邮通知管理员。发送有时间间隔,每小时只会发送一次。
4.将数值写入log。
5.设置crontab 每30秒运行一次。
servermonitor.sh
#!/bin/bash # 系统监控,记录cpu、memory、load average,当超过规定数值时发电邮通知管理员 # *** config start *** # 当前目录路径 root=$(cd "$(dirname "$0")"; pwd) # 当前服务器名 host=$(hostname) # log 文件路径 cpu_log="${root}/logs/cpu.log" mem_log="${root}/logs/mem.log" load_log="${root}/logs/load.log" # 通知电邮列表 notice_email='admin@admin.com' # cpu,memory,load average 记录上一次发送通知电邮时间 cpu_remark='/tmp/servermonitor_cpu.remark' mem_remark='/tmp/servermonitor_mem.remark' load_remark='/tmp/servermonitor_loadaverage.remark' # 发通知电邮间隔时间 remark_expire=3600 now=$(date +%s) # *** config end *** # *** function start *** # 获取cpu占用 function getcpu() { cpufree=$(vmstat 1 5 |sed -n '3,$p' |awk '{x = x + $15} end {print x/5}' |awk -f. '{print $1}') cpuused=$((100 - $cpufree)) echo $cpuused local remark remark=$(getremark ${cpu_remark}) # 检查cpu占用是否超过90% if [ "$remark" = "" ] && [ "$cpuused" -gt 90 ]; then echo "subject: ${host} cpu uses more than 90% $(date +%y-%m-%d' '%h:%m:%s)" | sendmail ${notice_email} echo "$(date +%s)" > "$cpu_remark" fi } # 获取内存使用情况 function getmem() { mem=$(free -m | sed -n '3,3p') used=$(echo $mem | awk -f ' ' '{print $3}') free=$(echo $mem | awk -f ' ' '{print $4}') total=$(($used + $free)) limit=$(($total/10)) echo "${total} ${used} ${free}" local remark remark=$(getremark ${mem_remark}) # 检查内存占用是否超过90% if [ "$remark" = "" ] && [ "$limit" -gt "$free" ]; then echo "subject: ${host} memory uses more than 90% $(date +%y-%m-%d' '%h:%m:%s)" | sendmail ${notice_email} echo "$(date +%s)" > "$mem_remark" fi } # 获取load average function getload() { load=$(uptime | awk -f 'load average: ' '{print $2}') m1=$(echo $load | awk -f ', ' '{print $1}') m5=$(echo $load | awk -f ', ' '{print $2}') m15=$(echo $load | awk -f ', ' '{print $3}') echo "${m1} ${m5} ${m15}" m1u=$(echo $m1 | awk -f '.' '{print $1}') local remark remark=$(getremark ${load_remark}) # 检查是否负载是否有压力 if [ "$remark" = "" ] && [ "$m1u" -gt "2" ]; then echo "subject: ${host} load average more than 2 $(date +%y-%m-%d' '%h:%m:%s)" | sendmail ${notice_email} echo "$(date +%s)" > "$load_remark" fi } # 获取上一次发送电邮时间 function getremark() { local remark if [ -f "$1" ] && [ -s "$1" ]; then remark=$(cat $1) if [ $(( $now - $remark )) -gt "$remark_expire" ]; then rm -f $1 remark="" fi else remark="" fi echo $remark } # *** function end *** cpuinfo=$(getcpu) meminfo=$(getmem) loadinfo=$(getload) echo "cpu: ${cpuinfo}" >> "${cpu_log}" echo "mem: ${meminfo}" >> "${mem_log}" echo "load: ${loadinfo}" >> "${load_log}" exit 0
监控网站是否异常
shell 监控网站是否异常的脚本,如有异常自动发电邮通知管理员。
流程:
1.检查网站返回的http_code是否等于200,如不是200视为异常。
2.检查网站的访问时间,超过maxloadtime(10秒)视为异常。
3.发送通知电邮后,在/tmp/monitor_load.remark记录发送时间,在一小时内不重复发送,如一小时后则清空/tmp/monitor_load.remark。
#!/bin/bash sites=("http://web01.example.com" "http://web02.example.com") # 要监控的网站 notice_email='me@example.com' # 管理员电邮 maxloadtime=10 # 访问超时时间设置 remarkfile='/tmp/monitor_load.remark' # 记录时否发送过通知电邮,如发送过则一小时内不再发送 issend=0 # 是否有发送电邮 expire=3600 # 每次发送电邮的间隔秒数 now=$(date +%s) if [ -f "$remarkfile" ] && [ -s "$remarkfile" ]; then remark=$(cat $remarkfile) # 删除过期的电邮发送时间记录文件 if [ $(( $now - $remark )) -gt "$expire" ]; then rm -f ${remarkfile} remark="" fi else remark="" fi # 循环判断每个site for site in ${sites[*]}; do printf "start to load ${site}\n" site_load_time=$(curl -o /dev/null -s -w "time_connect: %{time_connect}\ntime_starttransfer: %{time_starttransfer}\ntime_total: %{time_total}" "${site}") site_access=$(curl -o /dev/null -s -w %{http_code} "${site}") time_total=${site_load_time##*:} printf "$(date '+%y-%m-%d %h:%m:%s')\n" printf "site load time\n${site_load_time}\n" printf "site access:${site_access}\n\n" # not send if [ "$remark" = "" ]; then # check access if [ "$time_total" = "0.000" ] || [ "$site_access" != "200" ]; then echo "subject: ${site} can access $(date +%y-%m-%d' '%h:%m:%s)" | sendmail ${notice_email} issend=1 else # check load time if [ "${time_total%%.*}" -ge ${maxloadtime} ]; then echo "subject: ${site} load time total:${time_total} $(date +%y-%m-%d' '%h:%m:%s)" | sendmail ${notice_email} issend=1 fi fi fi done # 发送电邮后记录发送时间 if [ "$issend" = "1" ]; then echo "$(date +%s)" > $remarkfile fi exit 0