#!/usr/bin/env bash
PATH=/bin:/sbin:/usr/bin:/usr/sbin:/usr/local/bin:/usr/local/sbin:~/bin
export PATH
###########################################################
## Author        : Saiwa
## Email         : 93959@163.com
## Create Time   : 2023-06-03 15:00
## Last modified : 2023-06-11 18:30
## Filename      : ceph_tools.sh
## Description   : ceph常见报错处理等
###########################################################

ceph_tool_version="V20240320.02"
: '
V20230721.01 添加对报警信息mute的支持，减少告警频率和持续时间 @Peter提供代码
V20230721.02 mute优化，调到到自动处理报警之后，防止遗漏处理
V20230821.02 重启osd优化，hostname对应IP从远程文件获取，id_rsa不存在时手动输入密码
V20240320.02 增加通过传入参数自动处理报警，不再需要交互式选择
'

# Font Colors
RED='\033[0;31m'
BLUE='\033[1;34m'
GREEN='\033[0;32m'
YELLOW='\033[0;33m'
SKYBLUE='\033[0;36m'
SKYBLUE_B='\033[1;36m'
PURPLE_F='\033[5;35m'
PURPLE='\033[0;35m'
PLAIN='\033[0m'

# message lables
Tip="${GREEN}[注意]${PLAIN}"
Info="${GREEN}[信息]${PLAIN}"
Error="${RED}[错误]${PLAIN}"
Notice="${BLUE}[提示]${PLAIN}"
Notice2="${PURPLE}[提示]${PLAIN}"

add_timestamp() {
  while IFS= read -r line; do
    if [[ -n "$line" ]]; then
      echo "[$(date +"%Y-%m-%d %H:%M:%S")] $line"
    else
      echo "$line"
    fi
  done
}

osd_down(){
  echo -e "\n osd down，确认不是整个节点down了，登陆到对应存储节点启动即可\n"
  echo -e " $(ceph osd tree down)\n"
}

daemons_have_recently_crashed(){
  for id in $(ceph crash ls-new | awk 'NR!=1 {print $1}'); do ceph crash archive "$id"; done
}

not_scrubbed_in_time(){
#  ceph health detail | awk '/not scrubbed since/ {print $2}' | while read line; do ceph pg scrub $line; done
  for pg in $(ceph health detail | awk '/not scrubbed since/ {print $2}'); do ceph pg scrub "$pg"; done
}

not_deep-scrubbed_in_time(){
#  ceph health detail | awk '/not deep-scrubbed since/ {print $2}' | while read line; do ceph pg deep-scrub $line; done
  for pg in $(ceph health detail | awk '/not deep-scrubbed since/ {print $2}'); do ceph pg deep-scrub "$pg"; done
}

repair_pgs_not_scrubbed_deep-scrubbed(){
#  ceph health detail | awk '/not .*scrubbed since/ {print $2}' | while read line; do ceph pg repair $line; done
  for pg in $(ceph health detail | awk '/not .*scrubbed since/ {print $2}'); do ceph pg repair "$pg"; done
}

scrub_errors_pgs_inconsistent(){
  for pool in $(rados lspools)
  do
    # shellcheck disable=SC2207
    inconsistent_pgs=($(rados list-inconsistent-pg "$pool" | tr -d '[]"' | tr ',' ' '))
    for pgname in "${inconsistent_pgs[@]}"; do
      if [[ -n "$pgname" ]]; then
  #      echo "Repair pg: $pgname"
        ceph pg repair "$pgname"
      fi
    done
  done
}

Too_many_repaired_reads_on_1_osd(){
  how_repair() {
    echo && echo -e "
  #关闭ceph自带数据恢复平衡功能
  ceph osd set noout
  ceph osd set nobackfill
  ceph osd set norecover
  #查看异常osd ID
  ceph health detail
  #定位异常osd所在的节点
  ceph osd tree
  #登录对应的存储节点重启OSD
  systemctl restart ceph-osd@122       # @后面这个是osd号
  systemctl status ceph-osd@122        # 查看osd服务状态
  #开启ceph自带数据恢复平衡功能
  ceph osd unset noout
  ceph osd unset nobackfill
  ceph osd unset norecover" && echo
  }
  # 定义关闭数据平衡功能
  disable_data_balancing() {
    ceph osd set noout
#    ceph osd set nodown
    ceph osd set norecover
    ceph osd set nobackfill
  }
  # 定义开启数据平衡功能
  enable_data_balancing() {
    ceph osd unset noout
#    ceph osd unset nodown
    ceph osd unset norecover
    ceph osd unset nobackfill
  }
  # 获取异常osd，并尝试重启
  restart_osd() {
    for osd_id in $(ceph health detail | grep -Eo "osd.* had .* reads repaired" | awk -F '[. ]+' '{print $2}')
    do
      hostname=$(ceph osd find "$osd_id" | jq -r '.host')
      hostip=$(curl --max-time 5 -s http://137.175.79.80/tools/ceph_hosts | grep -oP -m1 '\d+\.\d+\.\d+\.\d+(?=.*'" $hostname"')')
      if [ -n "$hostip" ]; then
        echo " 检测到问题 osd $osd_id 位于 $hostname $hostip"
        ssh -i id_rsa "$hostip" "systemctl restart ceph-osd@$osd_id && echo $osd_id restart succeed || echo $osd_id restart failed"
        sleep 3s
      else
        echo " 无法解析问题 osd $osd_id 所在节点 ${hostname}对应IP地址，请手动处理"
        echo -e " 开启ceph自带数据恢复平衡功能"
        enable_data_balancing
        exit 1
      fi
      unset hostname hostip
    done
  }
  echo -e " 处理【Too many repaired xx】需要重启对应osd"
  echo -e " 检查执行脚本目录，是否有存储节点ssh私钥文件"
  # shellcheck disable=SC2015
  [ -f "$(pwd)/id_rsa" ] && echo -e " 私钥存在，稍后重启osd无需输入密码" || echo -e " 私钥不存在，稍后重启osd需要输入对应存储节点root的密码"
  echo -e " 关闭ceph自带数据恢复平衡功能"
  disable_data_balancing
  echo -e " 开始重启问题osd，如有多个，每隔3秒重启一个"
  restart_osd
  echo -e " 问题osd重启完毕，120秒后开启ceph自带数据恢复平衡功能"
  i=0
  while ((i < 120)); do
    echo -ne " 剩余 $((120 - i))s，请耐心稍等\r"
    sleep 1s
    ((i++))
  done
  echo -e " 开启ceph自带数据恢复平衡功能"
  enable_data_balancing
}

pg_recovery_unfound(){
  for pg in $(ceph pg ls recovery_unfound -f json | jq -r '.pg_stats[].pgid')
  do
    echo -e "$pg recovery_unfound, 尝试回退旧版!"
    ceph pg "$pg" mark_unfound_lost revert
    echo -e "\n如果回退旧版失败，可尝试删除，直接执行如下命令：\n"
    echo -e "ceph pg $pg mark_unfound_lost delete\n"
  done
}

auto_repair() {
  ceph_health=$(ceph health detail)
  i=0
  if echo "$ceph_health" | grep -Eq 'HEALTH_OK' && [ $debug == 0 ]; then echo -e "\n HEALTH_OK\n" && exit; fi
  if echo "$ceph_health" | grep -Eq 'osd down|osds down'; then osd_down; ((i++)); fi
  if echo "$ceph_health" | grep -Eq 'daemons have recently crashed'; then daemons_have_recently_crashed; ((i++)); fi
  if echo "$ceph_health" | grep -Eq 'OSD_SCRUB_ERRORS.*scrub error'; then scrub_errors_pgs_inconsistent; ((i++)); fi
  if echo "$ceph_health" | grep -Eq 'not deep-scrubbed since'; then not_deep-scrubbed_in_time; ((i++)); fi
  if echo "$ceph_health" | grep -Eq 'not scrubbed since'; then not_scrubbed_in_time; ((i++)); fi
  if echo "$ceph_health" | grep -Eq 'pg recovery_unfound'; then pg_recovery_unfound; ((i++)); fi
  if echo "$ceph_health" | grep -Eq 'Too many repaired reads on'; then Too_many_repaired_reads_on_1_osd; ((i++)); fi
  if [ "$i" -eq 0 ]; then
    echo -e "\n ${Notice2}未知报警信息，请手动处理...\n"
  else
    echo -e "\n ${Notice}自动处理了【$i】个报警信息！！！\n"
  fi
}


get_alert_code_and_mute() {
  output=$(ceph -s)
  if echo "$output" | grep -qi "health_ok"; then
      echo "Ceph healthy is OK"
  else
      echo "Ceph healthy is not OK"
      output=$(ceph health detail)
      ceph_alert_codes=$(echo "$output" | grep -o '\[\(ERR\|WRN\)\] [A-Z_]*' | awk '{ print $2 }')
      echo "Ceph Error Codes: $ceph_alert_codes"
      for alert_code in $ceph_alert_codes; do
          ceph health mute "$alert_code"
      done
  fi

  unset output ceph_alert_codes alert_code
}

xzcz(){
    read -r -p " 请选择[1|q]:" num
    echo
    case "$num" in
        0)
            start_menu
            ;;
        1)
            auto_repair
            get_alert_code_and_mute
            ;;
        q|Q)
            echo && exit
            ;;
        *)
            echo -e " ${Error}:输入有误，请重新输入！"
            xzcz
            ;;
    esac
}

start_menu(){
  echo -e "\n   ${SKYBLUE_B}CEPH TOOLS${PLAIN} ${RED}[${ceph_tool_version}]${PLAIN}
     -- ${PURPLE_F}By Saiwa${PLAIN} --
${SKYBLUE}—————CEPH常见报警问题修复———————${PLAIN}
 ${GREEN}1.${PLAIN} 自动处理常见报警
 ${GREEN}q.${PLAIN} 退出脚本
${SKYBLUE}————————————————————————————————${PLAIN}
 ${RED}★★${PLAIN} ${Tip}使用说明：
 ${RED}1.${PLAIN} 请在计算节点上执行本脚本！！！
 ${RED}2.${PLAIN} 涉及重启osd，请确保执行脚本的计算节点/etc/hosts有存储节点ip和hostname对应关系！！！" && echo
    echo -e " ${Info}当前系统版本：${YELLOW}$(sed 's/Linux release //g' /etc/redhat-release)${PLAIN}\n"
    xzcz
}

debug=0; if [ $# -ne 0 ]; then debug=$1; fi
# 检测新版本
echo -e "\n 检测版本...\c"
latest_version=$(curl --max-time 5 -s http://137.175.79.80/tools/ceph_tools.sh | grep -oP -m1 'ceph_tool_version="\K[^"]+')
if [[ -z "$latest_version" ]]; then
  echo -e " 检测新版本失败，请查检网络是否正常！！！"
elif [ "$ceph_tool_version" != "$latest_version" ]; then
  echo -e " 检测到新版本，将自动更新!!！下载新版本..."
  rm -rf "$0" || rm -rf ceph_tools.sh
  curl -O http://137.175.79.80/tools/ceph_tools.sh
  exec bash ceph_tools.sh $debug
else
  echo -e " 当前脚本为最新版本！！！"
fi

if [ $debug -eq 1 ]; then
  auto_repair
else
  start_menu
fi
