tg_monitor_bot/monitor.py

334 lines
14 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import paramiko
import time
import re
import os
from collections import defaultdict
import logging
logger = logging.getLogger(__name__)
# Глобальный словарь для хранения предыдущих значений RX/TX
prev_net_stats = {}
def get_ssh_client(host_config):
"""Создает SSH подключение через ssh-agent или ключ"""
client = paramiko.SSHClient()
client.set_missing_host_key_policy(paramiko.AutoAddPolicy())
try:
logger.info(f"🔌 Подключаюсь к {host_config['ip']}:{host_config['port']} как {host_config['user']}")
# Сначала пробуем ssh-agent
try:
from paramiko.agent import Agent
agent = Agent()
if agent.get_keys():
logger.info(f"🔑 Использую ssh-agent для {host_config['ip']}")
client.connect(
hostname=host_config['ip'],
port=host_config['port'],
username=host_config['user'],
timeout=10,
allow_agent=True,
look_for_keys=True
)
logger.info(f"✅ Подключено к {host_config['ip']} через ssh-agent")
return client
except Exception as e:
logger.debug(f"ssh-agent не доступен: {e}")
# Если ssh-agent не работает, пробуем через ключ
if os.path.exists(host_config['key_path']):
logger.info(f"🔑 Использую ключ: {host_config['key_path']}")
client.connect(
hostname=host_config['ip'],
port=host_config['port'],
username=host_config['user'],
key_filename=host_config['key_path'],
timeout=10
)
logger.info(f"✅ Подключено к {host_config['ip']} по ключу")
return client
else:
return f"❌ Файл ключа не найден: {host_config['key_path']}"
except paramiko.AuthenticationException:
error_msg = f"❌ Ошибка аутентификации для {host_config['ip']}! Проверь логин, ключ или ssh-agent."
logger.error(error_msg)
return error_msg
except paramiko.SSHException as e:
error_msg = f"❌ SSH ошибка для {host_config['ip']}: {str(e)}"
logger.error(error_msg)
return error_msg
except Exception as e:
error_msg = f"❌ Неизвестная ошибка для {host_config['ip']}: {str(e)}"
logger.error(error_msg)
return error_msg
def exec_cmd(client, command):
"""Выполняет команду и возвращает вывод"""
try:
stdin, stdout, stderr = client.exec_command(command, timeout=10)
return stdout.read().decode('utf-8').strip()
except Exception as e:
return f"CMD Error: {str(e)}"
def detect_host_type(client):
"""Определяет тип хоста: physical, vm, lxc, container"""
# Проверка на LXC
lxc_check = exec_cmd(client, "cat /proc/1/environ 2>/dev/null | tr '\\0' '\\n' | grep -i lxc || echo ''")
if lxc_check:
return "lxc"
# Проверка на контейнер (Docker и др)
container_check = exec_cmd(client, "cat /proc/1/environ 2>/dev/null | tr '\\0' '\\n' | grep -i container || echo ''")
if container_check:
return "container"
# Проверка на виртуальную машину
virt_check = exec_cmd(client, "systemd-detect-virt 2>/dev/null || echo 'none'")
if virt_check and virt_check not in ['none', '']:
return "vm"
# Проверка через dmidecode (если есть)
virt_check2 = exec_cmd(client, "dmidecode -s system-manufacturer 2>/dev/null || echo 'Unknown'")
if virt_check2:
virt_vendors = ['vmware', 'virtualbox', 'kvm', 'qemu', 'xen', 'microsoft', 'parallels']
if any(vendor in virt_check2.lower() for vendor in virt_vendors):
return "vm"
# Если ничего не найдено - считаем физическим железом
return "physical"
def check_alerts(metrics, host_type):
"""Проверяет критические метрики и возвращает список алертов"""
alerts = []
# CPU алерт (>80%)
cpu = metrics.get('cpu', 0)
if cpu > 80:
alerts.append(f"🔴 ВЫСОКАЯ ЗАГРУЗКА CPU: {cpu:.1f}%")
elif cpu > 60:
alerts.append(f"🟡 ПОВЫШЕННАЯ ЗАГРУЗКА CPU: {cpu:.1f}%")
# RAM алерт (>85%)
mem_percent = metrics.get('mem_percent', 0)
if mem_percent > 85:
alerts.append(f"🔴 НЕХВАТКА ОЗУ: {mem_percent:.1f}% (использовано {metrics.get('mem_used_mb', 0)} MB)")
elif mem_percent > 70:
alerts.append(f"🟡 ПОВЫШЕННОЕ ПОТРЕБЛЕНИЕ ОЗУ: {mem_percent:.1f}%")
# Диск алерт (>80%)
disk_usage = metrics.get('disk_usage', 0)
if disk_usage > 80:
alerts.append(f"🔴 МАЛО МЕСТА НА ДИСКЕ: {disk_usage:.1f}%")
elif disk_usage > 70:
alerts.append(f"🟡 ЗАКАНЧИВАЕТСЯ МЕСТО НА ДИСКЕ: {disk_usage:.1f}%")
# OOM алерт
oom = metrics.get('oom', '')
if oom and "Нет OOM" not in oom:
alerts.append(f"💀 OOM KILLER СРАБОТАЛ! Проверь логи!")
# Load Average алерт (если > количество ядер)
load_1 = float(metrics.get('load_1', 0))
if load_1 > 4:
alerts.append(f"🔴 ВЫСОКИЙ LOAD AVERAGE: {load_1}")
elif load_1 > 2:
alerts.append(f"🟡 ПОВЫШЕННЫЙ LOAD AVERAGE: {load_1}")
return alerts
def collect_metrics(host_config):
"""Сбор всех метрик с одного хоста"""
client = get_ssh_client(host_config)
if isinstance(client, str):
return {"error": client, "host": host_config['name'], "ip": host_config['ip']}
result = {
"host": host_config['name'],
"ip": host_config['ip'],
"metrics": {}
}
# 1. Определяем тип хоста
host_type = detect_host_type(client)
result["host_type"] = host_type
result["metrics"]["host_type"] = host_type
# 2. CPU загрузка (%)
cpu_output = exec_cmd(client, "top -bn1 | grep 'Cpu(s)' | awk '{print $2}' | cut -d'%' -f1")
try:
result["metrics"]["cpu"] = float(cpu_output.replace(',', '.'))
except:
result["metrics"]["cpu"] = 0.0
# 3. ОЗУ (используется в MB и %)
mem_output = exec_cmd(client, "free -m | grep Mem")
parts = re.split(r'\s+', mem_output)
if len(parts) >= 3:
total_mem = int(parts[1])
used_mem = int(parts[2])
result["metrics"]["mem_used_mb"] = used_mem
result["metrics"]["mem_percent"] = round((used_mem / total_mem) * 100, 1)
result["metrics"]["mem_total_mb"] = total_mem
else:
result["metrics"]["mem_used_mb"] = 0
result["metrics"]["mem_percent"] = 0
result["metrics"]["mem_total_mb"] = 0
# 4. Uptime и Load Average
uptime_output = exec_cmd(client, "uptime")
load_avg = re.search(r'load average:\s+([\d.]+),\s+([\d.]+),\s+([\d.]+)', uptime_output)
if load_avg:
result["metrics"]["load_1"] = load_avg.group(1)
result["metrics"]["load_5"] = load_avg.group(2)
result["metrics"]["load_15"] = load_avg.group(3)
uptime_seconds = exec_cmd(client, "cat /proc/uptime | awk '{print $1}'")
try:
uptime_seconds = float(uptime_seconds.split('.')[0])
days = uptime_seconds // 86400
hours = (uptime_seconds % 86400) // 3600
minutes = (uptime_seconds % 3600) // 60
result["metrics"]["uptime"] = f"{int(days)}d {int(hours)}h {int(minutes)}m"
except:
result["metrics"]["uptime"] = "N/A"
# 5. Сеть (трафик за последний час)
net_output = exec_cmd(client, "cat /proc/net/dev | tail -n +3 | grep -v lo")
current_net = {}
for line in net_output.split('\n'):
if not line.strip():
continue
parts = re.split(r'\s+', line.strip())
iface = parts[0].replace(':', '')
rx_bytes = int(parts[1])
tx_bytes = int(parts[9])
current_net[iface] = {'rx': rx_bytes, 'tx': tx_bytes}
global prev_net_stats
host_key = host_config['ip']
total_rx_hour = 0
total_tx_hour = 0
if host_key in prev_net_stats:
old_data = prev_net_stats[host_key]
for iface, vals in current_net.items():
if iface in old_data:
delta_rx = vals['rx'] - old_data[iface]['rx']
delta_tx = vals['tx'] - old_data[iface]['tx']
if delta_rx > 0:
total_rx_hour += delta_rx
if delta_tx > 0:
total_tx_hour += delta_tx
result["metrics"]["net_rx_mb"] = round(total_rx_hour / (1024 * 1024), 2)
result["metrics"]["net_tx_mb"] = round(total_tx_hour / (1024 * 1024), 2)
prev_net_stats[host_key] = current_net
# 6. Диски (iostat %util)
iostat_output = exec_cmd(client, "iostat -x -k 1 2 | grep -E 'sd[a-z]|vd[a-z]|nvme' | tail -n +3")
disk_stats = []
for line in iostat_output.split('\n'):
if not line.strip():
continue
parts = re.split(r'\s+', line.strip())
if len(parts) >= 14:
disk_name = parts[0]
try:
util = float(parts[-1].replace(',', '.'))
rps = float(parts[3].replace(',', '.'))
wps = float(parts[4].replace(',', '.'))
disk_stats.append({
"name": disk_name,
"util": util,
"rps": rps,
"wps": wps
})
except:
pass
disk_stats.sort(key=lambda x: x['util'], reverse=True)
result["metrics"]["disks"] = disk_stats[:3]
# 7. Свободное место на диске (%)
disk_usage_output = exec_cmd(client, "df -h / | tail -1 | awk '{print $5}' | sed 's/%//'")
try:
result["metrics"]["disk_usage"] = float(disk_usage_output.replace(',', '.'))
except:
result["metrics"]["disk_usage"] = 0
# 8. Температуры (ТОЛЬКО ДЛЯ ФИЗИЧЕСКИХ ХОСТОВ!)
temps = []
if host_type == "physical":
temp_output = exec_cmd(client, "sensors -u 2>/dev/null | grep -E 'temp[0-9]+_input|Core' | head -5")
if temp_output and "Error" not in temp_output and "command not found" not in temp_output:
for line in temp_output.split('\n'):
if '_input:' in line:
val = re.search(r'([\d.]+)', line)
if val:
temps.append(float(val.group(1)))
else:
# Если sensors не установлен - пробуем через /sys/class/thermal
temp_sys = exec_cmd(client, "cat /sys/class/thermal/thermal_zone*/temp 2>/dev/null | head -3")
if temp_sys and "No such file" not in temp_sys:
for line in temp_sys.split('\n'):
try:
temps.append(float(line) / 1000)
except:
pass
result["metrics"]["temps"] = temps
# 9. Топ-5 процессов по CPU и RAM
top_cpu = exec_cmd(client, "ps aux --sort=-%cpu | head -6 | tail -5")
top_mem = exec_cmd(client, "ps aux --sort=-%mem | head -6 | tail -5")
result["metrics"]["top_cpu"] = top_cpu
result["metrics"]["top_mem"] = top_mem
# 10. Логи (ошибки)
last_logs = exec_cmd(client, "journalctl -p 3 -xn 5 --no-pager 2>/dev/null || dmesg | tail -5")
result["metrics"]["logs_errors"] = last_logs
# 11. Fail2ban
fail2ban_status = exec_cmd(client, "fail2ban-client status 2>/dev/null")
if not fail2ban_status or "not found" in fail2ban_status:
ps_fail2ban = exec_cmd(client, "ps aux | grep -v grep | grep fail2ban-server")
if ps_fail2ban:
fail2ban_status = "✅ Fail2ban запущен"
else:
fail2ban_status = "❌ Fail2ban не установлен или не запущен"
else:
lines = fail2ban_status.split('\n')
jails_info = []
for line in lines:
if '|- Number of jail:' in line or '`- Number of jail:' in line:
jails_info.append(line.strip())
if 'Currently banned:' in line:
jails_info.append(line.strip())
if jails_info:
fail2ban_status = "✅ Fail2ban активен\n" + "\n".join(jails_info)
fail2ban_logs = exec_cmd(client, "tail -20 /var/log/fail2ban.log 2>/dev/null | grep -E 'BAN|UNBAN' | tail -5 || echo 'Нет записей о банах'")
active_bans = exec_cmd(client, "fail2ban-client status 2>/dev/null | grep -A 5 'Currently banned:'")
if active_bans and "Currently banned:" in active_bans:
fail2ban_logs += "\n\n🔒 Активные баны:\n" + active_bans
result["metrics"]["fail2ban_status"] = fail2ban_status
result["metrics"]["fail2ban_logs"] = fail2ban_logs
# 12. OOM
oom_check = exec_cmd(client, "dmesg | grep -i 'out of memory' | tail -3 || echo 'Нет OOM-событий'")
if oom_check and "Нет OOM-событий" not in oom_check:
result["metrics"]["oom"] = oom_check
else:
result["metrics"]["oom"] = "✅ Нет OOM-событий"
# 13. Проверка алертов
result["metrics"]["alerts"] = check_alerts(result["metrics"], host_type)
client.close()
return result