mirror of
https://github.com/EDeev/alertbot.git
synced 2026-10-08 04:59:29 +03:00
Compare commits
6 commits
1a77cf8620
...
6296fe9123
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6296fe9123 | ||
|
|
391900570a | ||
|
|
56638f5878 | ||
|
|
4a77957e93 | ||
|
|
c7d4142438 | ||
|
|
63c9976888 |
21 changed files with 1128 additions and 568 deletions
8
.dockerignore
Normal file
8
.dockerignore
Normal file
|
|
@ -0,0 +1,8 @@
|
||||||
|
.git
|
||||||
|
.github
|
||||||
|
**/__pycache__
|
||||||
|
.env
|
||||||
|
tests
|
||||||
|
*.md
|
||||||
|
*.db
|
||||||
|
*.log
|
||||||
|
|
@ -4,7 +4,8 @@ BOT_TOKEN=123456:your-telegram-bot-token
|
||||||
# Comma-separated Telegram user ids allowed to use the bot
|
# Comma-separated Telegram user ids allowed to use the bot
|
||||||
ALLOWED_IDS=111111111,222222222
|
ALLOWED_IDS=111111111,222222222
|
||||||
|
|
||||||
# Prometheus/Alertmanager endpoints (behind your own token-auth reverse-proxy path)
|
# Prometheus/Alertmanager endpoints behind your own reverse proxy.
|
||||||
|
# The bot sends "Authorization: Bearer <token>" — the proxy must check it.
|
||||||
AM_ALERTS_URL=https://your-domain/ambot/api/v2/alerts
|
AM_ALERTS_URL=https://your-domain/ambot/api/v2/alerts
|
||||||
PROM_QUERY_URL=https://your-domain/prombot/api/v1/query
|
PROM_QUERY_URL=https://your-domain/prombot/api/v1/query
|
||||||
AM_TOKEN=change-me
|
AM_TOKEN=change-me
|
||||||
|
|
@ -15,3 +16,8 @@ SERVERS_ORDER=srv1,srv2,srv3
|
||||||
|
|
||||||
ALERT_POLL_SECONDS=45
|
ALERT_POLL_SECONDS=45
|
||||||
ALERT_HTTP_TIMEOUT=15
|
ALERT_HTTP_TIMEOUT=15
|
||||||
|
# Report "monitoring unreachable" after this many failed polls in a row
|
||||||
|
ALERT_WATCHDOG_FAILURES=4
|
||||||
|
|
||||||
|
# SQLite file with subscriptions and alert state
|
||||||
|
ALERTBOT_DB=notifications.db
|
||||||
|
|
|
||||||
1
.gitattributes
vendored
Normal file
1
.gitattributes
vendored
Normal file
|
|
@ -0,0 +1 @@
|
||||||
|
* text=auto eol=lf
|
||||||
22
.github/workflows/ci.yml
vendored
Normal file
22
.github/workflows/ci.yml
vendored
Normal file
|
|
@ -0,0 +1,22 @@
|
||||||
|
name: CI
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
pull_request:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.12"
|
||||||
|
cache: pip
|
||||||
|
cache-dependency-path: requirements-dev.txt
|
||||||
|
- run: pip install -r requirements-dev.txt
|
||||||
|
- name: Ruff
|
||||||
|
run: ruff check .
|
||||||
|
- name: Тесты
|
||||||
|
run: pytest -q
|
||||||
42
.github/workflows/docker.yml
vendored
Normal file
42
.github/workflows/docker.yml
vendored
Normal file
|
|
@ -0,0 +1,42 @@
|
||||||
|
name: Docker
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags: ["v*"]
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
image:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
packages: write
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: docker/setup-buildx-action@v3
|
||||||
|
- uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: ghcr.io
|
||||||
|
username: ${{ github.actor }}
|
||||||
|
password: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
- uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: dcr.deev.su
|
||||||
|
username: ${{ secrets.ZOT_USERNAME }}
|
||||||
|
password: ${{ secrets.ZOT_PASSWORD }}
|
||||||
|
- id: meta
|
||||||
|
uses: docker/metadata-action@v5
|
||||||
|
with:
|
||||||
|
images: |
|
||||||
|
ghcr.io/edeev/alertbot
|
||||||
|
dcr.deev.su/edeev/alertbot
|
||||||
|
tags: |
|
||||||
|
type=semver,pattern={{version}}
|
||||||
|
type=semver,pattern={{major}}.{{minor}}
|
||||||
|
type=raw,value=latest
|
||||||
|
- uses: docker/build-push-action@v6
|
||||||
|
with:
|
||||||
|
context: .
|
||||||
|
push: true
|
||||||
|
tags: ${{ steps.meta.outputs.tags }}
|
||||||
|
labels: ${{ steps.meta.outputs.labels }}
|
||||||
17
Dockerfile
Normal file
17
Dockerfile
Normal file
|
|
@ -0,0 +1,17 @@
|
||||||
|
FROM python:3.12-slim
|
||||||
|
|
||||||
|
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||||
|
ALERTBOT_LOG_FILE= \
|
||||||
|
PYTHONUNBUFFERED=1 \
|
||||||
|
ALERTBOT_DB=/data/notifications.db
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
COPY requirements.txt .
|
||||||
|
RUN pip install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
|
COPY *.py ./
|
||||||
|
RUN useradd --create-home --uid 1000 app && mkdir -p /data && chown -R app:app /app /data
|
||||||
|
USER app
|
||||||
|
VOLUME ["/data"]
|
||||||
|
|
||||||
|
CMD ["python", "bot.py"]
|
||||||
109
README.en.md
Normal file
109
README.en.md
Normal file
|
|
@ -0,0 +1,109 @@
|
||||||
|
# AlertBot
|
||||||
|
|
||||||
|
[Русский](README.md) · **English**
|
||||||
|
|
||||||
|
[](https://github.com/EDeev/alertbot/actions/workflows/ci.yml)
|
||||||
|
[](https://github.com/EDeev/alertbot/actions/workflows/docker.yml)
|
||||||
|
[](https://github.com/EDeev/alertbot/releases)
|
||||||
|
[](LICENSE)
|
||||||
|
|
||||||
|
A Telegram bot for monitoring a small server fleet on top of Prometheus and Alertmanager: it sends
|
||||||
|
"problem / resolved" alerts and, on command, shows the state of servers, services and TLS certificates.
|
||||||
|
|
||||||
|
**Status:** personal project, in production · watches the author's five servers
|
||||||
|
|
||||||
|
```text
|
||||||
|
Серверы
|
||||||
|
|
||||||
|
SPB
|
||||||
|
CPU 7% · RAM 41% · диск 38% · swap 0%
|
||||||
|
load 0.21 · аптайм 12д 4ч
|
||||||
|
|
||||||
|
DORM · ⚠ диск
|
||||||
|
CPU 18% · RAM 63% · диск 87% · swap 2% · temp 52°C
|
||||||
|
load 1.40 · аптайм 3д 9ч
|
||||||
|
```
|
||||||
|
|
||||||
|
<sub>Sample <code>/servers</code> reply (illustrative values; the bot speaks Russian).</sub>
|
||||||
|
|
||||||
|
**Stack:** Python 3.10+ · aiogram 3 · aiohttp · SQLite · Prometheus HTTP API · Alertmanager API v2 · Docker
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
- **Alerts:** polls Alertmanager every N seconds, deduplicates by fingerprint and immediately notifies
|
||||||
|
subscribers about new and resolved alerts. No rules of its own — whatever Alertmanager has
|
||||||
|
- **Watchdog:** if Alertmanager itself stops responding, the bot reports it once and again when
|
||||||
|
monitoring is back
|
||||||
|
- **`/servers`** — CPU, RAM, disk, swap, load, uptime and temperature per node, ⚠ above thresholds
|
||||||
|
- **`/services`** — which Prometheus targets are down
|
||||||
|
- **`/certs`** — days until TLS certificates expire, warning under 14 days
|
||||||
|
- **`/alerts`** — active alerts; `/alerts on|off` — subscription
|
||||||
|
- Access is limited to a list of Telegram IDs; everyone else is silently ignored
|
||||||
|
|
||||||
|
## Quick start
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git clone https://github.com/EDeev/alertbot.git && cd alertbot
|
||||||
|
cp .env.example .env # bot token, IDs, Prometheus/Alertmanager URLs and tokens
|
||||||
|
docker compose up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
Prebuilt image: `docker pull ghcr.io/edeev/alertbot` or `docker pull dcr.deev.su/edeev/alertbot`.
|
||||||
|
|
||||||
|
## Installing without Docker
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python -m venv .venv && source .venv/bin/activate
|
||||||
|
pip install -r requirements.txt
|
||||||
|
cp .env.example .env
|
||||||
|
python bot.py
|
||||||
|
```
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
| Variable | Purpose |
|
||||||
|
|---|---|
|
||||||
|
| `BOT_TOKEN` | bot token from @BotFather |
|
||||||
|
| `ALLOWED_IDS` | comma-separated Telegram IDs allowed to use the bot |
|
||||||
|
| `AM_ALERTS_URL`, `PROM_QUERY_URL` | Alertmanager `/api/v2/alerts` and Prometheus `/api/v1/query` |
|
||||||
|
| `AM_TOKEN`, `PROM_TOKEN` | tokens sent in the `Authorization: Bearer` header |
|
||||||
|
| `SERVERS_ORDER` | comma-separated node names as in the metrics' `server` label |
|
||||||
|
| `ALERT_POLL_SECONDS`, `ALERT_HTTP_TIMEOUT` | poll interval and request timeout |
|
||||||
|
| `ALERT_WATCHDOG_FAILURES` | failed polls in a row before reporting monitoring as unreachable (default 4) |
|
||||||
|
| `ALERTBOT_DB`, `ALERTBOT_LOG_FILE` | SQLite file and log file (empty means stdout only) |
|
||||||
|
|
||||||
|
> [!IMPORTANT]
|
||||||
|
> Do not expose Prometheus and Alertmanager to the internet without authentication. The bot expects a
|
||||||
|
> reverse proxy that checks the `Authorization: Bearer …` token; see the nginx example in
|
||||||
|
> [docs/deploy.md](docs/deploy.md) (in Russian).
|
||||||
|
|
||||||
|
## Deployment
|
||||||
|
|
||||||
|
The bot runs on a separate VPS as a systemd unit. Prometheus and Alertmanager live on another server
|
||||||
|
behind nginx, which lets the bot in only with a token. GitHub Actions builds the Docker image on every
|
||||||
|
`v*` tag and publishes it to GitHub Packages and to `dcr.deev.su`.
|
||||||
|
|
||||||
|
## Development
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install -r requirements-dev.txt
|
||||||
|
ruff check . && pytest
|
||||||
|
```
|
||||||
|
|
||||||
|
The tests need no network: Telegram and Prometheus are faked. They cover alert deduplication,
|
||||||
|
resolved notifications, the watchdog and report texts.
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT — see [LICENSE](LICENSE).
|
||||||
|
|
||||||
|
## Author
|
||||||
|
|
||||||
|
**Egor Deev** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
<div align="center">
|
||||||
|
<sub>⭐ If you find this project useful, give it a star on GitHub!</sub>
|
||||||
|
<p><sub>Made with ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
|
||||||
|
</div>
|
||||||
149
README.md
149
README.md
|
|
@ -1,84 +1,109 @@
|
||||||
# AlertBot
|
# AlertBot
|
||||||
|
|
||||||
Telegram-бот для мониторинга парка серверов поверх Prometheus + Alertmanager. Работает как
|
**Русский** · [English](README.en.md)
|
||||||
push-уведомитель («что-то сломалось / починилось») и как справочная панель по команде
|
|
||||||
(`/servers`, `/services`, `/certs`).
|
[](https://github.com/EDeev/alertbot/actions/workflows/ci.yml)
|
||||||
|
[](https://github.com/EDeev/alertbot/actions/workflows/docker.yml)
|
||||||
|
[](https://github.com/EDeev/alertbot/releases)
|
||||||
|
[](LICENSE)
|
||||||
|
|
||||||
|
Telegram-бот для мониторинга парка серверов поверх Prometheus и Alertmanager: присылает алерты
|
||||||
|
«проблема / в норме» и по команде показывает состояние серверов, сервисов и TLS-сертификатов.
|
||||||
|
|
||||||
|
**Статус:** личный проект, работает · следит за пятью серверами автора
|
||||||
|
|
||||||
|
```text
|
||||||
|
Серверы
|
||||||
|
|
||||||
|
SPB
|
||||||
|
CPU 7% · RAM 41% · диск 38% · swap 0%
|
||||||
|
load 0.21 · аптайм 12д 4ч
|
||||||
|
|
||||||
|
DORM · ⚠ диск
|
||||||
|
CPU 18% · RAM 63% · диск 87% · swap 2% · temp 52°C
|
||||||
|
load 1.40 · аптайм 3д 9ч
|
||||||
|
```
|
||||||
|
|
||||||
|
<sub>Пример ответа на <code>/servers</code> (значения условные).</sub>
|
||||||
|
|
||||||
|
**Стек:** Python 3.10+ · aiogram 3 · aiohttp · SQLite · Prometheus HTTP API · Alertmanager API v2 · Docker
|
||||||
|
|
||||||
## Возможности
|
## Возможности
|
||||||
|
|
||||||
- **Алерты в реальном времени** — фоновая задача раз в N секунд опрашивает Alertmanager,
|
- **Алерты:** раз в N секунд опрашивает Alertmanager, дедуплицирует по fingerprint и сразу пишет
|
||||||
дедуплицирует по fingerprint и сразу шлёт «Проблема · …» / «В норме · …» подписанным
|
«Проблема · …» и «В норме · …» подписчикам. Своих правил нет — всё, что настроено в Alertmanager
|
||||||
пользователям. Никаких собственных правил — подхватывает всё, что уже настроено в
|
- **Сторож:** если сам Alertmanager перестал отвечать, бот один раз сообщает об этом и ещё раз —
|
||||||
Alertmanager, по лейблам, а не по именам целей.
|
когда мониторинг вернулся
|
||||||
- **`/servers`** — CPU, RAM, диск, swap, load, аптайм и (если есть hwmon-датчики)
|
- **`/servers`** — CPU, RAM, диск, swap, load, аптайм и температура по узлам, ⚠ при превышении порогов
|
||||||
температура по каждому узлу, с ⚠ при превышении порогов.
|
- **`/services`** — какие цели Prometheus не отвечают
|
||||||
- **`/services`** — какие цели Prometheus сейчас `up`, какие нет.
|
- **`/certs`** — дни до истечения TLS-сертификатов, предупреждение меньше чем за 14 дней
|
||||||
- **`/certs`** — сколько дней осталось у каждого TLS-сертификата, с предупреждением при <14 дней.
|
- **`/alerts`** — активные алерты; `/alerts on|off` — подписка
|
||||||
- **`/alerts`** — список активных алертов сейчас; `/alerts on|off` — подписка/отписка.
|
- Доступ — только для Telegram ID из списка, остальные молча игнорируются
|
||||||
- Доступ — по вайтлисту Telegram user id, все остальные тихо игнорируются.
|
|
||||||
|
|
||||||
## Требования
|
## Быстрый старт
|
||||||
|
|
||||||
- Python 3.10+
|
|
||||||
- Свой Prometheus + Alertmanager с уже настроенными правилами алертов.
|
|
||||||
- Alertmanager и Prometheus должны быть доступны боту по HTTP — либо напрямую (если бот
|
|
||||||
крутится на той же машине), либо через reverse-proxy с токен-параметром в URL (так это
|
|
||||||
сделано в этом проекте — см. `AM_ALERTS_URL`/`PROM_QUERY_URL`/`AM_TOKEN`/`PROM_TOKEN`
|
|
||||||
в `.env.example`), чтобы не открывать сами Prometheus/Alertmanager в интернет без авторизации.
|
|
||||||
|
|
||||||
## Установка
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
git clone https://github.com/EDeev/alertbot.git && cd alertbot
|
||||||
|
cp .env.example .env # токен бота, ID, адреса и токены Prometheus/Alertmanager
|
||||||
|
docker compose up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
Готовый образ: `docker pull ghcr.io/edeev/alertbot` или `docker pull dcr.deev.su/edeev/alertbot`.
|
||||||
|
|
||||||
|
## Установка без Docker
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python -m venv .venv && source .venv/bin/activate
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
cp .env.example .env
|
cp .env.example .env
|
||||||
# заполнить .env своими значениями
|
|
||||||
python bot.py
|
python bot.py
|
||||||
```
|
```
|
||||||
|
|
||||||
## Переменные окружения
|
## Конфигурация
|
||||||
|
|
||||||
Все — в `.env.example`:
|
| Переменная | Назначение |
|
||||||
|
|---|---|
|
||||||
|
| `BOT_TOKEN` | токен бота от @BotFather |
|
||||||
|
| `ALLOWED_IDS` | Telegram ID через запятую, кому можно пользоваться ботом |
|
||||||
|
| `AM_ALERTS_URL`, `PROM_QUERY_URL` | Alertmanager `/api/v2/alerts` и Prometheus `/api/v1/query` |
|
||||||
|
| `AM_TOKEN`, `PROM_TOKEN` | токены, бот передаёт их в заголовке `Authorization: Bearer` |
|
||||||
|
| `SERVERS_ORDER` | имена узлов через запятую, как в лейбле `server` метрик |
|
||||||
|
| `ALERT_POLL_SECONDS`, `ALERT_HTTP_TIMEOUT` | период опроса и таймаут запросов |
|
||||||
|
| `ALERT_WATCHDOG_FAILURES` | после скольких неудачных опросов подряд сообщать о недоступности (по умолчанию 4) |
|
||||||
|
| `ALERTBOT_DB`, `ALERTBOT_LOG_FILE` | файл SQLite и файл лога (пусто — только stdout) |
|
||||||
|
|
||||||
- `BOT_TOKEN` — токен бота от @BotFather.
|
> [!IMPORTANT]
|
||||||
- `ALLOWED_IDS` — Telegram user id через запятую, кому разрешено пользоваться ботом.
|
> Не открывайте Prometheus и Alertmanager в интернет без авторизации. Бот рассчитан на обратный прокси,
|
||||||
- `AM_ALERTS_URL` / `PROM_QUERY_URL` / `AM_TOKEN` / `PROM_TOKEN` — адреса и токены до
|
> который проверяет токен из заголовка `Authorization: Bearer …`, — пример для nginx в
|
||||||
Alertmanager API (`/api/v2/alerts`) и Prometheus API (`/api/v1/query`).
|
> [docs/deploy.md](docs/deploy.md).
|
||||||
- `SERVERS_ORDER` — имена узлов через запятую, ровно как они указаны в лейбле `server`
|
|
||||||
у твоих метрик (`node_exporter` и т.п.) — определяет порядок и состав вывода `/servers`.
|
|
||||||
- `ALERT_POLL_SECONDS` / `ALERT_HTTP_TIMEOUT` — период опроса и таймаут запросов.
|
|
||||||
|
|
||||||
## Структура
|
## Развёртывание
|
||||||
|
|
||||||
```
|
Бот работает на отдельном VPS как systemd-юнит. Prometheus и Alertmanager стоят на другом сервере
|
||||||
init.py # конфиг из переменных окружения, инициализация Bot/Dispatcher
|
за nginx, который пускает бота только с токеном. Docker-образ собирает GitHub Actions на каждый тег
|
||||||
sql.py # SQLite: подписчики на алерты, дедуп по fingerprint
|
`v*` и публикует в GitHub Packages и в реестр `dcr.deev.su`.
|
||||||
handlers.py # команды бота + фоновый опрос Alertmanager
|
|
||||||
bot.py # точка входа
|
## Разработка
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install -r requirements-dev.txt
|
||||||
|
ruff check . && pytest
|
||||||
```
|
```
|
||||||
|
|
||||||
Хранилище — один файл SQLite (`notifications.db`, путь по умолчанию задаётся при
|
Тесты работают без сети: Telegram и Prometheus подменяются. Проверяются дедупликация алертов,
|
||||||
инициализации `DatabaseManager`), без внешних зависимостей вроде Redis/Postgres.
|
сообщения о возврате в норму, сторож и тексты сводок.
|
||||||
|
|
||||||
## Продакшен
|
|
||||||
|
|
||||||
Юнит `systemd` — самый простой способ держать бота в фоне постоянно:
|
|
||||||
|
|
||||||
```ini
|
|
||||||
[Unit]
|
|
||||||
Description=AlertBot
|
|
||||||
After=network.target
|
|
||||||
|
|
||||||
[Service]
|
|
||||||
Type=simple
|
|
||||||
WorkingDirectory=/opt/alertbot
|
|
||||||
ExecStart=/opt/alertbot/venv/bin/python3 bot.py
|
|
||||||
Restart=always
|
|
||||||
RestartSec=10
|
|
||||||
|
|
||||||
[Install]
|
|
||||||
WantedBy=multi-user.target
|
|
||||||
```
|
|
||||||
|
|
||||||
## Лицензия
|
## Лицензия
|
||||||
|
|
||||||
MIT
|
MIT — см. [LICENSE](LICENSE).
|
||||||
|
|
||||||
|
## Автор
|
||||||
|
|
||||||
|
**Деев Егор Викторович** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
<div align="center">
|
||||||
|
<sub>⭐ Если проект оказался полезным, поставьте звёздочку на GitHub!</sub>
|
||||||
|
<p><sub>Сделано с ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
|
||||||
|
</div>
|
||||||
|
|
|
||||||
106
bot.py
106
bot.py
|
|
@ -1,51 +1,55 @@
|
||||||
import asyncio
|
import asyncio
|
||||||
import logging
|
import logging
|
||||||
import sys
|
import os
|
||||||
|
import sys
|
||||||
from aiogram.types import BotCommand
|
|
||||||
|
from aiogram.types import BotCommand
|
||||||
from init import bot, dp
|
|
||||||
from handlers import router, alert_poller
|
from init import bot, dp
|
||||||
|
from handlers import router, alert_poller
|
||||||
logging.basicConfig(
|
|
||||||
level=logging.INFO,
|
_log_handlers = [logging.StreamHandler(sys.stdout)]
|
||||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s',
|
# File log is optional: in Docker set ALERTBOT_LOG_FILE= (empty) and read `docker logs`
|
||||||
handlers=[logging.StreamHandler(sys.stdout),
|
if os.environ.get("ALERTBOT_LOG_FILE", "bot.log"):
|
||||||
logging.FileHandler('bot.log', encoding='utf-8')],
|
_log_handlers.append(logging.FileHandler(os.environ.get("ALERTBOT_LOG_FILE", "bot.log"), encoding="utf-8"))
|
||||||
)
|
logging.basicConfig(
|
||||||
log = logging.getLogger(__name__)
|
level=logging.INFO,
|
||||||
|
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s',
|
||||||
COMMANDS = [
|
handlers=_log_handlers,
|
||||||
BotCommand(command="servers", description="CPU / RAM / диск / аптайм по узлам"),
|
)
|
||||||
BotCommand(command="services", description="что работает и что нет"),
|
log = logging.getLogger(__name__)
|
||||||
BotCommand(command="certs", description="дни до истечения TLS"),
|
|
||||||
BotCommand(command="alerts", description="активные алерты (on/off — подписка)"),
|
COMMANDS = [
|
||||||
BotCommand(command="status", description="что включено"),
|
BotCommand(command="servers", description="CPU / RAM / диск / аптайм по узлам"),
|
||||||
BotCommand(command="help", description="справка"),
|
BotCommand(command="services", description="что работает и что нет"),
|
||||||
]
|
BotCommand(command="certs", description="дни до истечения TLS"),
|
||||||
|
BotCommand(command="alerts", description="активные алерты (on/off — подписка)"),
|
||||||
|
BotCommand(command="status", description="что включено"),
|
||||||
async def main() -> None:
|
BotCommand(command="help", description="справка"),
|
||||||
log.info("alertbot starting")
|
]
|
||||||
dp.include_router(router)
|
|
||||||
await bot.delete_webhook(drop_pending_updates=True)
|
|
||||||
try:
|
async def main() -> None:
|
||||||
await bot.set_my_commands(COMMANDS)
|
log.info("alertbot starting")
|
||||||
except Exception as e:
|
dp.include_router(router)
|
||||||
log.warning("set_my_commands failed: %s", e)
|
await bot.delete_webhook(drop_pending_updates=True)
|
||||||
|
try:
|
||||||
poller = asyncio.create_task(alert_poller())
|
await bot.set_my_commands(COMMANDS)
|
||||||
try:
|
except Exception as e:
|
||||||
await dp.start_polling(bot, allowed_updates=dp.resolve_used_update_types(),
|
log.warning("set_my_commands failed: %s", e)
|
||||||
timeout=20, relax=0.1)
|
|
||||||
finally:
|
poller = asyncio.create_task(alert_poller())
|
||||||
poller.cancel()
|
try:
|
||||||
await bot.session.close()
|
await dp.start_polling(bot, allowed_updates=dp.resolve_used_update_types(),
|
||||||
log.info("alertbot stopped")
|
timeout=20, relax=0.1)
|
||||||
|
finally:
|
||||||
|
poller.cancel()
|
||||||
if __name__ == "__main__":
|
await bot.session.close()
|
||||||
try:
|
log.info("alertbot stopped")
|
||||||
asyncio.run(main())
|
|
||||||
except (KeyboardInterrupt, SystemExit):
|
|
||||||
pass
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
asyncio.run(main())
|
||||||
|
except (KeyboardInterrupt, SystemExit):
|
||||||
|
pass
|
||||||
|
|
|
||||||
14
compose.yaml
Normal file
14
compose.yaml
Normal file
|
|
@ -0,0 +1,14 @@
|
||||||
|
services:
|
||||||
|
bot:
|
||||||
|
build: .
|
||||||
|
image: ghcr.io/edeev/alertbot:latest
|
||||||
|
env_file: .env
|
||||||
|
environment:
|
||||||
|
ALERTBOT_DB: /data/notifications.db
|
||||||
|
ALERTBOT_LOG_FILE: ""
|
||||||
|
volumes:
|
||||||
|
- data:/data
|
||||||
|
restart: unless-stopped
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
data:
|
||||||
58
docs/deploy.md
Normal file
58
docs/deploy.md
Normal file
|
|
@ -0,0 +1,58 @@
|
||||||
|
# Развёртывание
|
||||||
|
|
||||||
|
## Docker
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cp .env.example .env
|
||||||
|
docker compose up -d
|
||||||
|
docker compose logs -f bot
|
||||||
|
```
|
||||||
|
|
||||||
|
Подписки и состояние алертов хранятся в томе `data` (`/data/notifications.db`).
|
||||||
|
|
||||||
|
## systemd
|
||||||
|
|
||||||
|
```ini
|
||||||
|
[Unit]
|
||||||
|
Description=AlertBot
|
||||||
|
After=network.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=simple
|
||||||
|
WorkingDirectory=/opt/alertbot
|
||||||
|
ExecStart=/opt/alertbot/venv/bin/python bot.py
|
||||||
|
Restart=always
|
||||||
|
RestartSec=10
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
|
```
|
||||||
|
|
||||||
|
## Обратный прокси перед Prometheus и Alertmanager (nginx)
|
||||||
|
|
||||||
|
Бот отправляет токен в заголовке `Authorization: Bearer <токен>`. Пример проверки:
|
||||||
|
|
||||||
|
```nginx
|
||||||
|
# в контексте http (например, /etc/nginx/conf.d/alertbot-auth.conf)
|
||||||
|
map $http_authorization $alertbot_auth_ok {
|
||||||
|
default 0;
|
||||||
|
"Bearer ВАШ_ТОКЕН" 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
# в server { ... }
|
||||||
|
location /prombot/ {
|
||||||
|
if ($alertbot_auth_ok = 0) { return 403; }
|
||||||
|
proxy_pass http://127.0.0.1:9090/;
|
||||||
|
}
|
||||||
|
location /ambot/ {
|
||||||
|
if ($alertbot_auth_ok = 0) { return 403; }
|
||||||
|
proxy_pass http://127.0.0.1:9093/;
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Что ожидается от Prometheus
|
||||||
|
|
||||||
|
- у метрик node_exporter есть лейбл `server` с короткими именами узлов (как в `SERVERS_ORDER`);
|
||||||
|
- джобы node_exporter называются `node_*`;
|
||||||
|
- для `/certs` — blackbox exporter с метрикой `probe_ssl_earliest_cert_expiry`;
|
||||||
|
- температура берётся из `node_hwmon_temp_celsius` с `chip=~"pci.*"` (настоящие датчики).
|
||||||
709
handlers.py
709
handlers.py
|
|
@ -1,333 +1,376 @@
|
||||||
import asyncio
|
import asyncio
|
||||||
import html
|
import html
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import time
|
import time
|
||||||
from typing import Dict, List
|
from typing import Dict, List
|
||||||
|
|
||||||
import aiohttp
|
import aiohttp
|
||||||
from aiogram import BaseMiddleware, Router
|
from aiogram import BaseMiddleware, Router
|
||||||
from aiogram.filters import Command, CommandObject
|
from aiogram.filters import Command, CommandObject
|
||||||
from aiogram.types import Message
|
from aiogram.types import Message
|
||||||
|
|
||||||
from init import (ALERT_HTTP_TIMEOUT, ALERT_POLL_SECONDS, ALLOWED_IDS, AM_ALERTS_URL,
|
from init import (ALERT_HTTP_TIMEOUT, ALERT_POLL_SECONDS, ALERT_WATCHDOG_FAILURES, ALLOWED_IDS,
|
||||||
AM_TOKEN, PROM_QUERY_URL, PROM_TOKEN, bot)
|
AM_ALERTS_URL, AM_TOKEN, DB_PATH, PROM_QUERY_URL, PROM_TOKEN, bot)
|
||||||
from sql import DatabaseManager
|
from sql import DatabaseManager
|
||||||
|
|
||||||
router = Router()
|
router = Router()
|
||||||
db = DatabaseManager()
|
db = DatabaseManager(DB_PATH)
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
# Node names as used in Prometheus' "server" label — match your own scrape config.
|
# Node names as used in Prometheus' "server" label — match your own scrape config.
|
||||||
SERVERS_ORDER = [s for s in os.environ.get("SERVERS_ORDER", "srv1,srv2,srv3").split(",") if s]
|
SERVERS_ORDER = [s for s in os.environ.get("SERVERS_ORDER", "srv1,srv2,srv3").split(",") if s]
|
||||||
|
|
||||||
|
|
||||||
class AccessMiddleware(BaseMiddleware):
|
class AccessMiddleware(BaseMiddleware):
|
||||||
async def __call__(self, handler, event, data):
|
async def __call__(self, handler, event, data):
|
||||||
u = getattr(event, "from_user", None) or data.get("event_from_user")
|
u = getattr(event, "from_user", None) or data.get("event_from_user")
|
||||||
if u is None or u.id in ALLOWED_IDS:
|
if u is None or u.id in ALLOWED_IDS:
|
||||||
return await handler(event, data)
|
return await handler(event, data)
|
||||||
return None # silently ignore everyone else
|
return None # silently ignore everyone else
|
||||||
|
|
||||||
|
|
||||||
router.message.outer_middleware(AccessMiddleware())
|
router.message.outer_middleware(AccessMiddleware())
|
||||||
|
|
||||||
|
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
# small helpers
|
# small helpers
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
def fmt_uptime(sec: float) -> str:
|
def fmt_uptime(sec: float) -> str:
|
||||||
sec = int(max(0, sec))
|
sec = int(max(0, sec))
|
||||||
d, rem = divmod(sec, 86400)
|
d, rem = divmod(sec, 86400)
|
||||||
h = rem // 3600
|
h = rem // 3600
|
||||||
return f"{d}д {h}ч" if d else f"{h}ч"
|
return f"{d}д {h}ч" if d else f"{h}ч"
|
||||||
|
|
||||||
|
|
||||||
def settings_block(row) -> str:
|
def settings_block(row) -> str:
|
||||||
_, _u, alerts_on = row
|
_, _u, alerts_on = row
|
||||||
firing = len(db.list_firing_fingerprints())
|
firing = len(db.list_firing_fingerprints())
|
||||||
out = f"Алерты: {'включены' if alerts_on else 'выключены'}"
|
out = f"Алерты: {'включены' if alerts_on else 'выключены'}"
|
||||||
if firing:
|
if firing:
|
||||||
out += f"\nСейчас активных алертов: <b>{firing}</b>"
|
out += f"\nСейчас активных алертов: <b>{firing}</b>"
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
# commands
|
# commands
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
@router.message(Command("start"))
|
@router.message(Command("start"))
|
||||||
async def cmd_start(msg: Message):
|
async def cmd_start(msg: Message):
|
||||||
uid = msg.from_user.id
|
uid = msg.from_user.id
|
||||||
db.add_user(uid, msg.from_user.username or msg.from_user.first_name or str(uid))
|
db.add_user(uid, msg.from_user.username or msg.from_user.first_name or str(uid))
|
||||||
row = db.get_user_info(uid)
|
row = db.get_user_info(uid)
|
||||||
await msg.answer(
|
await msg.answer(
|
||||||
"<b>Мониторинг инфраструктуры</b>\n\n"
|
"<b>Мониторинг инфраструктуры</b>\n\n"
|
||||||
"• сразу сообщает о проблемах и о возврате в норму (алерты);\n"
|
"• сразу сообщает о проблемах и о возврате в норму (алерты);\n"
|
||||||
"• по запросу отдаёт состояние серверов, сервисов и сертификатов.\n\n"
|
"• по запросу отдаёт состояние серверов, сервисов и сертификатов.\n\n"
|
||||||
+ settings_block(row) + "\n\n"
|
+ settings_block(row) + "\n\n"
|
||||||
"Команды — в меню слева от поля ввода, или /help."
|
"Команды — в меню слева от поля ввода, или /help."
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("help"))
|
@router.message(Command("help"))
|
||||||
async def cmd_help(msg: Message):
|
async def cmd_help(msg: Message):
|
||||||
await msg.answer(
|
await msg.answer(
|
||||||
"<b>Команды</b>\n\n"
|
"<b>Команды</b>\n\n"
|
||||||
"<b>Сводки</b>\n"
|
"<b>Сводки</b>\n"
|
||||||
"/servers — CPU, RAM, диск, аптайм по узлам\n"
|
"/servers — CPU, RAM, диск, аптайм по узлам\n"
|
||||||
"/services — что работает и что нет\n"
|
"/services — что работает и что нет\n"
|
||||||
"/certs — сколько дней осталось у TLS-сертификатов\n"
|
"/certs — сколько дней осталось у TLS-сертификатов\n"
|
||||||
"/alerts — активные алерты сейчас\n\n"
|
"/alerts — активные алерты сейчас\n\n"
|
||||||
"<b>Настройки</b>\n"
|
"<b>Настройки</b>\n"
|
||||||
"/status — что включено\n"
|
"/status — что включено\n"
|
||||||
"/alerts on | off — подписка на алерты"
|
"/alerts on | off — подписка на алерты"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("status"))
|
@router.message(Command("status"))
|
||||||
async def cmd_status(msg: Message):
|
async def cmd_status(msg: Message):
|
||||||
row = db.get_user_info(msg.from_user.id)
|
row = db.get_user_info(msg.from_user.id)
|
||||||
if not row:
|
if not row:
|
||||||
await msg.answer("Нажми /start.")
|
await msg.answer("Нажми /start.")
|
||||||
return
|
return
|
||||||
await msg.answer("<b>Настройки</b>\n\n" + settings_block(row))
|
await msg.answer("<b>Настройки</b>\n\n" + settings_block(row))
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("alerts"))
|
@router.message(Command("alerts"))
|
||||||
async def cmd_alerts(msg: Message, command: CommandObject):
|
async def cmd_alerts(msg: Message, command: CommandObject):
|
||||||
uid = msg.from_user.id
|
uid = msg.from_user.id
|
||||||
if not db.get_user_info(uid):
|
if not db.get_user_info(uid):
|
||||||
db.add_user(uid, msg.from_user.username or str(uid))
|
db.add_user(uid, msg.from_user.username or str(uid))
|
||||||
arg = (command.args or "").strip().lower()
|
arg = (command.args or "").strip().lower()
|
||||||
if arg in ("on", "вкл"):
|
if arg in ("on", "вкл"):
|
||||||
db.set_alerts(uid, True)
|
db.set_alerts(uid, True)
|
||||||
await msg.answer("Алерты включены.")
|
await msg.answer("Алерты включены.")
|
||||||
return
|
return
|
||||||
if arg in ("off", "выкл"):
|
if arg in ("off", "выкл"):
|
||||||
db.set_alerts(uid, False)
|
db.set_alerts(uid, False)
|
||||||
await msg.answer("Алерты выключены.")
|
await msg.answer("Алерты выключены.")
|
||||||
return
|
return
|
||||||
await msg.answer(await render_active_alerts())
|
await msg.answer(await render_active_alerts())
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("servers"))
|
@router.message(Command("servers"))
|
||||||
async def cmd_servers(msg: Message):
|
async def cmd_servers(msg: Message):
|
||||||
await msg.answer(await render_servers())
|
await msg.answer(await render_servers())
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("services"))
|
@router.message(Command("services"))
|
||||||
async def cmd_services(msg: Message):
|
async def cmd_services(msg: Message):
|
||||||
await msg.answer(await render_services())
|
await msg.answer(await render_services())
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("certs"))
|
@router.message(Command("certs"))
|
||||||
async def cmd_certs(msg: Message):
|
async def cmd_certs(msg: Message):
|
||||||
await msg.answer(await render_certs())
|
await msg.answer(await render_certs())
|
||||||
|
|
||||||
|
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
# Prometheus
|
# Prometheus
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
async def _promq(session: aiohttp.ClientSession, query: str) -> List[dict]:
|
def _auth(token: str) -> dict:
|
||||||
async with session.get(PROM_QUERY_URL, params={"query": query, "token": PROM_TOKEN},
|
"""Token goes in the Authorization header, not in the URL (URLs end up in proxy logs)."""
|
||||||
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
return {"Authorization": f"Bearer {token}"}
|
||||||
r.raise_for_status()
|
|
||||||
j = await r.json()
|
|
||||||
if j.get("status") != "success":
|
async def _promq(session: aiohttp.ClientSession, query: str) -> List[dict]:
|
||||||
raise RuntimeError(j.get("error", "prometheus error"))
|
async with session.get(PROM_QUERY_URL, params={"query": query},
|
||||||
return j["data"]["result"]
|
headers=_auth(PROM_TOKEN),
|
||||||
|
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
||||||
|
r.raise_for_status()
|
||||||
def _by_server(result: List[dict]) -> Dict[str, float]:
|
j = await r.json()
|
||||||
out = {}
|
if j.get("status") != "success":
|
||||||
for s in result:
|
raise RuntimeError(j.get("error", "prometheus error"))
|
||||||
srv = s["metric"].get("server")
|
return j["data"]["result"]
|
||||||
if srv:
|
|
||||||
try:
|
|
||||||
out[srv] = float(s["value"][1])
|
def _by_server(result: List[dict]) -> Dict[str, float]:
|
||||||
except (ValueError, TypeError):
|
out = {}
|
||||||
pass
|
for s in result:
|
||||||
return out
|
srv = s["metric"].get("server")
|
||||||
|
if srv:
|
||||||
|
try:
|
||||||
async def render_servers() -> str:
|
out[srv] = float(s["value"][1])
|
||||||
q = {
|
except (ValueError, TypeError):
|
||||||
"cpu": '100 - (avg by (server) (rate(node_cpu_seconds_total{mode="idle"}[2m])) * 100)',
|
pass
|
||||||
"ram": '(1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) * 100',
|
return out
|
||||||
"disk": '(1 - avg by (server)(node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}) '
|
|
||||||
'/ avg by (server)(node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"})) * 100',
|
|
||||||
"load": 'node_load1',
|
async def render_servers() -> str:
|
||||||
"up": 'up{job=~"node_.*"}',
|
q = {
|
||||||
"boot": 'node_boot_time_seconds',
|
"cpu": '100 - (avg by (server) (rate(node_cpu_seconds_total{mode="idle"}[2m])) * 100)',
|
||||||
"swap": '(node_memory_SwapTotal_bytes > bool 0) * (1 - node_memory_SwapFree_bytes / (node_memory_SwapTotal_bytes > 0)) * 100',
|
"ram": '(1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) * 100',
|
||||||
# real hwmon sensors only (chip=~"pci.*") — excludes bogus acpitz/thermal_zone
|
"disk": '(1 - avg by (server)(node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}) '
|
||||||
# readings some laptops-as-servers report as a stuck fake value
|
'/ avg by (server)(node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"})) * 100',
|
||||||
"temp": 'max by (server) (node_hwmon_temp_celsius{chip=~"pci.*"})',
|
"load": 'node_load1',
|
||||||
}
|
"up": 'up{job=~"node_.*"}',
|
||||||
try:
|
"boot": 'node_boot_time_seconds',
|
||||||
async with aiohttp.ClientSession() as s:
|
"swap": '(node_memory_SwapTotal_bytes > bool 0) * (1 - node_memory_SwapFree_bytes / (node_memory_SwapTotal_bytes > 0)) * 100',
|
||||||
res = {k: _by_server(await _promq(s, v)) for k, v in q.items()}
|
# real hwmon sensors only (chip=~"pci.*") — excludes bogus acpitz/thermal_zone
|
||||||
except Exception as e:
|
# readings some laptops-as-servers report as a stuck fake value
|
||||||
return f"Не удалось получить метрики: {e}"
|
"temp": 'max by (server) (node_hwmon_temp_celsius{chip=~"pci.*"})',
|
||||||
|
}
|
||||||
now = time.time()
|
try:
|
||||||
out = ["<b>Серверы</b>"]
|
async with aiohttp.ClientSession() as s:
|
||||||
for srv in SERVERS_ORDER:
|
res = {k: _by_server(await _promq(s, v)) for k, v in q.items()}
|
||||||
if srv not in res["up"]:
|
except Exception as e:
|
||||||
out.append(f"\n<b>{srv.upper()}</b> — нет данных")
|
return f"Не удалось получить метрики: {e}"
|
||||||
continue
|
|
||||||
alive = res["up"].get(srv, 0) >= 1
|
now = time.time()
|
||||||
cpu, ram = res["cpu"].get(srv), res["ram"].get(srv)
|
out = ["<b>Серверы</b>"]
|
||||||
disk, swap = res["disk"].get(srv), res["swap"].get(srv, 0.0)
|
for srv in SERVERS_ORDER:
|
||||||
load = res["load"].get(srv)
|
if srv not in res["up"]:
|
||||||
temp = res["temp"].get(srv)
|
out.append(f"\n<b>{srv.upper()}</b> — нет данных")
|
||||||
up = fmt_uptime(now - res["boot"][srv]) if srv in res["boot"] else "?"
|
continue
|
||||||
|
alive = res["up"].get(srv, 0) >= 1
|
||||||
def g(x):
|
cpu, ram = res["cpu"].get(srv), res["ram"].get(srv)
|
||||||
return f"{x:.0f}%" if isinstance(x, (int, float)) else "?"
|
disk, swap = res["disk"].get(srv), res["swap"].get(srv, 0.0)
|
||||||
|
load = res["load"].get(srv)
|
||||||
warn = []
|
temp = res["temp"].get(srv)
|
||||||
if isinstance(disk, float) and disk >= 85:
|
up = fmt_uptime(now - res["boot"][srv]) if srv in res["boot"] else "?"
|
||||||
warn.append("диск")
|
|
||||||
if isinstance(ram, float) and ram >= 90:
|
def g(x):
|
||||||
warn.append("память")
|
return f"{x:.0f}%" if isinstance(x, (int, float)) else "?"
|
||||||
if isinstance(swap, float) and swap >= 60:
|
|
||||||
warn.append("swap")
|
warn = []
|
||||||
if isinstance(temp, float) and temp >= 85:
|
if isinstance(disk, float) and disk >= 85:
|
||||||
warn.append("температура")
|
warn.append("диск")
|
||||||
head = f"\n<b>{srv.upper()}</b>"
|
if isinstance(ram, float) and ram >= 90:
|
||||||
if not alive:
|
warn.append("память")
|
||||||
head += " · НЕ ОТВЕЧАЕТ"
|
if isinstance(swap, float) and swap >= 60:
|
||||||
elif warn:
|
warn.append("swap")
|
||||||
head += " · ⚠ " + ", ".join(warn)
|
if isinstance(temp, float) and temp >= 85:
|
||||||
out.append(head)
|
warn.append("температура")
|
||||||
line = f"CPU {g(cpu)} · RAM {g(ram)} · диск {g(disk)} · swap {g(swap)}"
|
head = f"\n<b>{srv.upper()}</b>"
|
||||||
if isinstance(temp, float):
|
if not alive:
|
||||||
line += f" · temp {temp:.0f}°C"
|
head += " · НЕ ОТВЕЧАЕТ"
|
||||||
out.append(line)
|
elif warn:
|
||||||
out.append(f"load {load:.2f} · аптайм {up}" if isinstance(load, float)
|
head += " · ⚠ " + ", ".join(warn)
|
||||||
else f"аптайм {up}")
|
out.append(head)
|
||||||
return "\n".join(out)
|
line = f"CPU {g(cpu)} · RAM {g(ram)} · диск {g(disk)} · swap {g(swap)}"
|
||||||
|
if isinstance(temp, float):
|
||||||
|
line += f" · temp {temp:.0f}°C"
|
||||||
async def render_services() -> str:
|
out.append(line)
|
||||||
try:
|
out.append(f"load {load:.2f} · аптайм {up}" if isinstance(load, float)
|
||||||
async with aiohttp.ClientSession() as s:
|
else f"аптайм {up}")
|
||||||
res = await _promq(s, "up")
|
return "\n".join(out)
|
||||||
except Exception as e:
|
|
||||||
return f"Не удалось получить статус сервисов: {e}"
|
|
||||||
total = len(res)
|
async def render_services() -> str:
|
||||||
down = [(m["metric"].get("job", "?"), m["metric"].get("instance", "?"))
|
try:
|
||||||
for m in res if m["value"][1] != "1"]
|
async with aiohttp.ClientSession() as s:
|
||||||
if not down:
|
res = await _promq(s, "up")
|
||||||
return f"<b>Сервисы</b>\n\nВсе {total} в норме."
|
except Exception as e:
|
||||||
lines = [f"<b>Сервисы</b>\n\n{total - len(down)}/{total} в норме. Не отвечают:"]
|
return f"Не удалось получить статус сервисов: {e}"
|
||||||
for job, inst in sorted(down):
|
total = len(res)
|
||||||
lines.append(f"• {html.escape(job)} — {html.escape(inst)}")
|
down = [(m["metric"].get("job", "?"), m["metric"].get("instance", "?"))
|
||||||
return "\n".join(lines)
|
for m in res if m["value"][1] != "1"]
|
||||||
|
if not down:
|
||||||
|
return f"<b>Сервисы</b>\n\nВсе {total} в норме."
|
||||||
async def render_certs() -> str:
|
lines = [f"<b>Сервисы</b>\n\n{total - len(down)}/{total} в норме. Не отвечают:"]
|
||||||
try:
|
for job, inst in sorted(down):
|
||||||
async with aiohttp.ClientSession() as s:
|
lines.append(f"• {html.escape(job)} — {html.escape(inst)}")
|
||||||
res = await _promq(s, "(probe_ssl_earliest_cert_expiry - time()) / 86400")
|
return "\n".join(lines)
|
||||||
except Exception as e:
|
|
||||||
return f"Не удалось получить данные по сертификатам: {e}"
|
|
||||||
rows = []
|
async def render_certs() -> str:
|
||||||
for m in res:
|
try:
|
||||||
try:
|
async with aiohttp.ClientSession() as s:
|
||||||
days = float(m["value"][1])
|
res = await _promq(s, "(probe_ssl_earliest_cert_expiry - time()) / 86400")
|
||||||
except (ValueError, TypeError):
|
except Exception as e:
|
||||||
continue
|
return f"Не удалось получить данные по сертификатам: {e}"
|
||||||
host = m["metric"].get("instance", "?").replace("https://", "").split("/")[0]
|
rows = []
|
||||||
rows.append((days, host))
|
for m in res:
|
||||||
if not rows:
|
try:
|
||||||
return "Данных по сертификатам нет."
|
days = float(m["value"][1])
|
||||||
rows.sort()
|
except (ValueError, TypeError):
|
||||||
soon = [f"{h} — {d:.0f} дн" for d, h in rows if d < 14]
|
continue
|
||||||
body = "\n".join(f"{d:>4.0f} {h}" for d, h in rows)
|
host = m["metric"].get("instance", "?").replace("https://", "").split("/")[0]
|
||||||
head = "<b>TLS-сертификаты</b>"
|
rows.append((days, host))
|
||||||
if soon:
|
if not rows:
|
||||||
head += "\n⚠ скоро истекают: " + "; ".join(soon)
|
return "Данных по сертификатам нет."
|
||||||
return head + f"\n<pre>{body}</pre>"
|
rows.sort()
|
||||||
|
soon = [f"{h} — {d:.0f} дн" for d, h in rows if d < 14]
|
||||||
|
body = "\n".join(f"{d:>4.0f} {h}" for d, h in rows)
|
||||||
# ======================================================================
|
head = "<b>TLS-сертификаты</b>"
|
||||||
# Alertmanager
|
if soon:
|
||||||
# ======================================================================
|
head += "\n⚠ скоро истекают: " + "; ".join(soon)
|
||||||
def _sev(labels): return labels.get("severity", "").lower()
|
return head + f"\n<pre>{body}</pre>"
|
||||||
def _where(labels): return labels.get("server") or labels.get("instance") or labels.get("job") or ""
|
|
||||||
|
|
||||||
|
# ======================================================================
|
||||||
def _fmt_firing(a: dict) -> str:
|
# Alertmanager
|
||||||
lb, an = a.get("labels", {}), a.get("annotations", {})
|
# ======================================================================
|
||||||
name = html.escape(lb.get("alertname", "alert"))
|
def _sev(labels): return labels.get("severity", "").lower()
|
||||||
sub = " · ".join(x for x in (_sev(lb), html.escape(_where(lb))) if x)
|
def _where(labels): return labels.get("server") or labels.get("instance") or labels.get("job") or ""
|
||||||
summ = html.escape(an.get("summary") or an.get("description") or "")
|
|
||||||
text = f"<b>Проблема · {name}</b>"
|
|
||||||
if sub:
|
def _fmt_firing(a: dict) -> str:
|
||||||
text += f"\n{sub}"
|
lb, an = a.get("labels", {}), a.get("annotations", {})
|
||||||
if summ:
|
name = html.escape(lb.get("alertname", "alert"))
|
||||||
text += f"\n{summ}"
|
sub = " · ".join(x for x in (_sev(lb), html.escape(_where(lb))) if x)
|
||||||
return text
|
summ = html.escape(an.get("summary") or an.get("description") or "")
|
||||||
|
text = f"<b>Проблема · {name}</b>"
|
||||||
|
if sub:
|
||||||
def _fmt_resolved(name: str) -> str:
|
text += f"\n{sub}"
|
||||||
return f"<b>В норме · {html.escape(name)}</b>"
|
if summ:
|
||||||
|
text += f"\n{summ}"
|
||||||
|
return text
|
||||||
async def _fetch_alerts(session: aiohttp.ClientSession) -> List[dict]:
|
|
||||||
params = {"token": AM_TOKEN, "active": "true", "silenced": "false", "inhibited": "false"}
|
|
||||||
async with session.get(AM_ALERTS_URL, params=params,
|
def _fmt_resolved(name: str) -> str:
|
||||||
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
return f"<b>В норме · {html.escape(name)}</b>"
|
||||||
r.raise_for_status()
|
|
||||||
return await r.json()
|
|
||||||
|
async def _fetch_alerts(session: aiohttp.ClientSession) -> List[dict]:
|
||||||
|
params = {"active": "true", "silenced": "false", "inhibited": "false"}
|
||||||
async def _broadcast(text: str):
|
async with session.get(AM_ALERTS_URL, params=params, headers=_auth(AM_TOKEN),
|
||||||
for uid in db.get_alert_users():
|
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
||||||
try:
|
r.raise_for_status()
|
||||||
await bot.send_message(uid, text)
|
return await r.json()
|
||||||
except Exception as e:
|
|
||||||
log.error("alert to %s failed: %s", uid, e)
|
|
||||||
|
async def _broadcast(text: str):
|
||||||
|
for uid in db.get_alert_users():
|
||||||
async def render_active_alerts() -> str:
|
try:
|
||||||
try:
|
await bot.send_message(uid, text)
|
||||||
async with aiohttp.ClientSession() as s:
|
except Exception as e:
|
||||||
data = await _fetch_alerts(s)
|
log.error("alert to %s failed: %s", uid, e)
|
||||||
except Exception as e:
|
|
||||||
return f"Не удалось получить алерты: {e}"
|
|
||||||
firing = [a for a in data if a.get("status", {}).get("state") == "active"]
|
async def render_active_alerts() -> str:
|
||||||
if not firing:
|
try:
|
||||||
return "Активных алертов нет."
|
async with aiohttp.ClientSession() as s:
|
||||||
return f"<b>Активные алерты: {len(firing)}</b>\n\n" + "\n\n".join(_fmt_firing(a) for a in firing)
|
data = await _fetch_alerts(s)
|
||||||
|
except Exception as e:
|
||||||
|
return f"Не удалось получить алерты: {e}"
|
||||||
async def alert_poller():
|
firing = [a for a in data if a.get("status", {}).get("state") == "active"]
|
||||||
log.info("alert poller started (every %ss)", ALERT_POLL_SECONDS)
|
if not firing:
|
||||||
async with aiohttp.ClientSession() as session:
|
return "Активных алертов нет."
|
||||||
while True:
|
return f"<b>Активные алерты: {len(firing)}</b>\n\n" + "\n\n".join(_fmt_firing(a) for a in firing)
|
||||||
try:
|
|
||||||
data = await _fetch_alerts(session)
|
|
||||||
current = {}
|
def _fmt_watchdog_down(error: str) -> str:
|
||||||
for a in data:
|
return ("<b>Мониторинг недоступен</b>\nAlertmanager не отвечает, о новых проблемах бот сейчас не узнает.\n"
|
||||||
if a.get("status", {}).get("state") != "active":
|
f"Последняя ошибка: {html.escape(error)}")
|
||||||
continue
|
|
||||||
fp = a.get("fingerprint")
|
|
||||||
if not fp:
|
def _fmt_watchdog_up() -> str:
|
||||||
continue
|
return "<b>Мониторинг снова доступен</b>"
|
||||||
current[fp] = a
|
|
||||||
if db.get_alert_status(fp) != "firing":
|
|
||||||
await _broadcast(_fmt_firing(a))
|
async def process_alerts(data: List[dict]) -> None:
|
||||||
db.upsert_alert(fp, "firing", a.get("labels", {}).get("alertname", "alert"))
|
"""One poll: announce new firing alerts and alerts that have resolved since the last poll."""
|
||||||
for fp, name in db.list_firing_fingerprints():
|
current = {}
|
||||||
if fp not in current:
|
for a in data:
|
||||||
await _broadcast(_fmt_resolved(name))
|
if a.get("status", {}).get("state") != "active":
|
||||||
db.upsert_alert(fp, "resolved", name)
|
continue
|
||||||
db.purge_old_resolved()
|
fp = a.get("fingerprint")
|
||||||
except asyncio.CancelledError:
|
if not fp:
|
||||||
break
|
continue
|
||||||
except Exception as e:
|
current[fp] = a
|
||||||
log.warning("alert poll failed: %s", e)
|
if db.get_alert_status(fp) != "firing":
|
||||||
await asyncio.sleep(ALERT_POLL_SECONDS)
|
await _broadcast(_fmt_firing(a))
|
||||||
|
db.upsert_alert(fp, "firing", a.get("labels", {}).get("alertname", "alert"))
|
||||||
|
for fp, name in db.list_firing_fingerprints():
|
||||||
|
if fp not in current:
|
||||||
|
await _broadcast(_fmt_resolved(name))
|
||||||
|
db.upsert_alert(fp, "resolved", name)
|
||||||
|
db.purge_old_resolved()
|
||||||
|
|
||||||
|
|
||||||
|
class Watchdog:
|
||||||
|
"""Counts failed polls in a row; reports once when monitoring is lost and once when it is back."""
|
||||||
|
|
||||||
|
def __init__(self, threshold: int):
|
||||||
|
self.threshold = threshold
|
||||||
|
self.failures = 0
|
||||||
|
self.reported = False
|
||||||
|
|
||||||
|
async def failed(self, error: str) -> None:
|
||||||
|
self.failures += 1
|
||||||
|
if self.failures >= self.threshold and not self.reported:
|
||||||
|
self.reported = True
|
||||||
|
await _broadcast(_fmt_watchdog_down(error))
|
||||||
|
|
||||||
|
async def ok(self) -> None:
|
||||||
|
if self.reported:
|
||||||
|
await _broadcast(_fmt_watchdog_up())
|
||||||
|
self.failures = 0
|
||||||
|
self.reported = False
|
||||||
|
|
||||||
|
|
||||||
|
async def alert_poller():
|
||||||
|
log.info("alert poller started (every %ss)", ALERT_POLL_SECONDS)
|
||||||
|
watchdog = Watchdog(ALERT_WATCHDOG_FAILURES)
|
||||||
|
async with aiohttp.ClientSession() as session:
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
await process_alerts(await _fetch_alerts(session))
|
||||||
|
await watchdog.ok()
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
break
|
||||||
|
except Exception as e:
|
||||||
|
log.warning("alert poll failed: %s", e)
|
||||||
|
await watchdog.failed(str(e) or type(e).__name__)
|
||||||
|
await asyncio.sleep(ALERT_POLL_SECONDS)
|
||||||
|
|
|
||||||
3
init.py
3
init.py
|
|
@ -20,6 +20,9 @@ AM_TOKEN = os.environ["AM_TOKEN"]
|
||||||
PROM_TOKEN = os.environ["PROM_TOKEN"]
|
PROM_TOKEN = os.environ["PROM_TOKEN"]
|
||||||
ALERT_POLL_SECONDS = int(os.environ.get("ALERT_POLL_SECONDS", "45"))
|
ALERT_POLL_SECONDS = int(os.environ.get("ALERT_POLL_SECONDS", "45"))
|
||||||
ALERT_HTTP_TIMEOUT = int(os.environ.get("ALERT_HTTP_TIMEOUT", "15"))
|
ALERT_HTTP_TIMEOUT = int(os.environ.get("ALERT_HTTP_TIMEOUT", "15"))
|
||||||
|
# After this many failed polls in a row the bot reports that monitoring itself is unreachable
|
||||||
|
ALERT_WATCHDOG_FAILURES = int(os.environ.get("ALERT_WATCHDOG_FAILURES", "4"))
|
||||||
|
DB_PATH = os.environ.get("ALERTBOT_DB", "notifications.db")
|
||||||
|
|
||||||
bot = Bot(token=BOT_TOKEN, default=DefaultBotProperties(parse_mode=ParseMode.HTML))
|
bot = Bot(token=BOT_TOKEN, default=DefaultBotProperties(parse_mode=ParseMode.HTML))
|
||||||
dp = Dispatcher(storage=MemoryStorage())
|
dp = Dispatcher(storage=MemoryStorage())
|
||||||
|
|
|
||||||
13
pyproject.toml
Normal file
13
pyproject.toml
Normal file
|
|
@ -0,0 +1,13 @@
|
||||||
|
[tool.ruff]
|
||||||
|
target-version = "py310"
|
||||||
|
line-length = 120
|
||||||
|
|
||||||
|
[tool.ruff.lint]
|
||||||
|
select = ["E", "F", "W", "B", "S"]
|
||||||
|
ignore = ["E501", "E701"]
|
||||||
|
|
||||||
|
[tool.ruff.lint.per-file-ignores]
|
||||||
|
"tests/*" = ["S101", "S105", "S106"]
|
||||||
|
|
||||||
|
[tool.pytest.ini_options]
|
||||||
|
testpaths = ["tests"]
|
||||||
3
requirements-dev.txt
Normal file
3
requirements-dev.txt
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
-r requirements.txt
|
||||||
|
pytest==8.4.2
|
||||||
|
ruff==0.14.0
|
||||||
|
|
@ -1,3 +1,3 @@
|
||||||
aiogram>=3.0
|
aiogram==3.31.0
|
||||||
aiohttp
|
aiohttp==3.14.3
|
||||||
python-dotenv
|
python-dotenv==1.2.1
|
||||||
|
|
|
||||||
236
sql.py
236
sql.py
|
|
@ -1,118 +1,118 @@
|
||||||
import sqlite3
|
import sqlite3
|
||||||
from typing import List, Optional, Tuple
|
from typing import List, Optional, Tuple
|
||||||
|
|
||||||
|
|
||||||
class DatabaseManager:
|
class DatabaseManager:
|
||||||
def __init__(self, db_path: str = "notifications.db"):
|
def __init__(self, db_path: str = "notifications.db"):
|
||||||
self.db_path = db_path
|
self.db_path = db_path
|
||||||
self.init_database()
|
self.init_database()
|
||||||
|
|
||||||
def _conn(self):
|
def _conn(self):
|
||||||
return sqlite3.connect(self.db_path)
|
return sqlite3.connect(self.db_path)
|
||||||
|
|
||||||
def init_database(self):
|
def init_database(self):
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
conn.execute('''
|
conn.execute('''
|
||||||
CREATE TABLE IF NOT EXISTS users (
|
CREATE TABLE IF NOT EXISTS users (
|
||||||
user_id INTEGER PRIMARY KEY,
|
user_id INTEGER PRIMARY KEY,
|
||||||
username TEXT,
|
username TEXT,
|
||||||
alerts_enabled BOOLEAN DEFAULT 1,
|
alerts_enabled BOOLEAN DEFAULT 1,
|
||||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
||||||
)
|
)
|
||||||
''')
|
''')
|
||||||
conn.execute('''
|
conn.execute('''
|
||||||
CREATE TABLE IF NOT EXISTS alert_seen (
|
CREATE TABLE IF NOT EXISTS alert_seen (
|
||||||
fingerprint TEXT PRIMARY KEY,
|
fingerprint TEXT PRIMARY KEY,
|
||||||
status TEXT,
|
status TEXT,
|
||||||
name TEXT,
|
name TEXT,
|
||||||
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
||||||
)
|
)
|
||||||
''')
|
''')
|
||||||
conn.commit()
|
conn.commit()
|
||||||
|
|
||||||
# ---------- users ----------
|
# ---------- users ----------
|
||||||
def add_user(self, user_id: int, username: str) -> bool:
|
def add_user(self, user_id: int, username: str) -> bool:
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"INSERT INTO users (user_id, username) VALUES (?, ?) "
|
"INSERT INTO users (user_id, username) VALUES (?, ?) "
|
||||||
"ON CONFLICT(user_id) DO UPDATE SET username=excluded.username",
|
"ON CONFLICT(user_id) DO UPDATE SET username=excluded.username",
|
||||||
(user_id, username),
|
(user_id, username),
|
||||||
)
|
)
|
||||||
conn.commit()
|
conn.commit()
|
||||||
return True
|
return True
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def set_alerts(self, user_id: int, value: bool) -> bool:
|
def set_alerts(self, user_id: int, value: bool) -> bool:
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
conn.execute("UPDATE users SET alerts_enabled = ? WHERE user_id = ?",
|
conn.execute("UPDATE users SET alerts_enabled = ? WHERE user_id = ?",
|
||||||
(1 if value else 0, user_id))
|
(1 if value else 0, user_id))
|
||||||
conn.commit()
|
conn.commit()
|
||||||
return True
|
return True
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def get_user_info(self, user_id: int) -> Optional[Tuple]:
|
def get_user_info(self, user_id: int) -> Optional[Tuple]:
|
||||||
"""(user_id, username, alerts_enabled)"""
|
"""(user_id, username, alerts_enabled)"""
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
return conn.execute(
|
return conn.execute(
|
||||||
"SELECT user_id, username, alerts_enabled FROM users WHERE user_id = ?",
|
"SELECT user_id, username, alerts_enabled FROM users WHERE user_id = ?",
|
||||||
(user_id,)
|
(user_id,)
|
||||||
).fetchone()
|
).fetchone()
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def get_alert_users(self) -> List[int]:
|
def get_alert_users(self) -> List[int]:
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
return [r[0] for r in conn.execute(
|
return [r[0] for r in conn.execute(
|
||||||
"SELECT user_id FROM users WHERE alerts_enabled = 1"
|
"SELECT user_id FROM users WHERE alerts_enabled = 1"
|
||||||
).fetchall()]
|
).fetchall()]
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
# ---------- alert dedup ----------
|
# ---------- alert dedup ----------
|
||||||
def get_alert_status(self, fingerprint: str) -> Optional[str]:
|
def get_alert_status(self, fingerprint: str) -> Optional[str]:
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
r = conn.execute("SELECT status FROM alert_seen WHERE fingerprint = ?",
|
r = conn.execute("SELECT status FROM alert_seen WHERE fingerprint = ?",
|
||||||
(fingerprint,)).fetchone()
|
(fingerprint,)).fetchone()
|
||||||
return r[0] if r else None
|
return r[0] if r else None
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def upsert_alert(self, fingerprint: str, status: str, name: str):
|
def upsert_alert(self, fingerprint: str, status: str, name: str):
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"INSERT INTO alert_seen (fingerprint, status, name, updated_at) "
|
"INSERT INTO alert_seen (fingerprint, status, name, updated_at) "
|
||||||
"VALUES (?, ?, ?, CURRENT_TIMESTAMP) "
|
"VALUES (?, ?, ?, CURRENT_TIMESTAMP) "
|
||||||
"ON CONFLICT(fingerprint) DO UPDATE SET status=excluded.status, "
|
"ON CONFLICT(fingerprint) DO UPDATE SET status=excluded.status, "
|
||||||
"name=excluded.name, updated_at=CURRENT_TIMESTAMP",
|
"name=excluded.name, updated_at=CURRENT_TIMESTAMP",
|
||||||
(fingerprint, status, name),
|
(fingerprint, status, name),
|
||||||
)
|
)
|
||||||
conn.commit()
|
conn.commit()
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def list_firing_fingerprints(self) -> List[Tuple[str, str]]:
|
def list_firing_fingerprints(self) -> List[Tuple[str, str]]:
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
return conn.execute(
|
return conn.execute(
|
||||||
"SELECT fingerprint, name FROM alert_seen WHERE status = 'firing'"
|
"SELECT fingerprint, name FROM alert_seen WHERE status = 'firing'"
|
||||||
).fetchall()
|
).fetchall()
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
def purge_old_resolved(self, days: int = 3):
|
def purge_old_resolved(self, days: int = 3):
|
||||||
try:
|
try:
|
||||||
with self._conn() as conn:
|
with self._conn() as conn:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"DELETE FROM alert_seen WHERE status = 'resolved' "
|
"DELETE FROM alert_seen WHERE status = 'resolved' "
|
||||||
"AND updated_at < datetime('now', ?)", (f'-{days} days',))
|
"AND updated_at < datetime('now', ?)", (f'-{days} days',))
|
||||||
conn.commit()
|
conn.commit()
|
||||||
except sqlite3.Error:
|
except sqlite3.Error:
|
||||||
pass
|
pass
|
||||||
|
|
|
||||||
0
tests/__init__.py
Normal file
0
tests/__init__.py
Normal file
62
tests/conftest.py
Normal file
62
tests/conftest.py
Normal file
|
|
@ -0,0 +1,62 @@
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
os.environ.update({
|
||||||
|
"BOT_TOKEN": "123456:TEST-token-for-unit-tests-only",
|
||||||
|
"ALLOWED_IDS": "1,2",
|
||||||
|
"AM_ALERTS_URL": "https://example.test/ambot/api/v2/alerts",
|
||||||
|
"PROM_QUERY_URL": "https://example.test/prombot/api/v1/query",
|
||||||
|
"AM_TOKEN": "am-token",
|
||||||
|
"PROM_TOKEN": "prom-token",
|
||||||
|
"SERVERS_ORDER": "spb,ams,dorm",
|
||||||
|
"ALERTBOT_DB": str(Path(tempfile.mkdtemp()) / "test.db"),
|
||||||
|
"ALERT_WATCHDOG_FAILURES": "3",
|
||||||
|
})
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
import pytest # noqa: E402
|
||||||
|
|
||||||
|
import handlers # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def fresh_db():
|
||||||
|
with handlers.db._conn() as conn:
|
||||||
|
conn.execute("DELETE FROM users")
|
||||||
|
conn.execute("DELETE FROM alert_seen")
|
||||||
|
handlers.db.add_user(1, "owner")
|
||||||
|
handlers.db.add_user(2, "friend")
|
||||||
|
handlers.db.set_alerts(2, False)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def sent(monkeypatch):
|
||||||
|
"""Messages the bot would send: (user_id, text)."""
|
||||||
|
out = []
|
||||||
|
|
||||||
|
async def fake_send(uid, text):
|
||||||
|
out.append((uid, text))
|
||||||
|
|
||||||
|
monkeypatch.setattr(handlers.bot, "send_message", fake_send)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def prom(monkeypatch):
|
||||||
|
"""Fake Prometheus: maps a substring of the query to a result vector."""
|
||||||
|
answers = {}
|
||||||
|
|
||||||
|
async def fake_promq(session, query):
|
||||||
|
for key, value in answers.items():
|
||||||
|
if key in query:
|
||||||
|
return value
|
||||||
|
return []
|
||||||
|
|
||||||
|
monkeypatch.setattr(handlers, "_promq", fake_promq)
|
||||||
|
return answers
|
||||||
|
|
||||||
|
|
||||||
|
def vec(**by_server):
|
||||||
|
return [{"metric": {"server": k}, "value": [0, str(v)]} for k, v in by_server.items()]
|
||||||
61
tests/test_alerts.py
Normal file
61
tests/test_alerts.py
Normal file
|
|
@ -0,0 +1,61 @@
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
import handlers
|
||||||
|
|
||||||
|
|
||||||
|
def alert(fp, name="NodeDown", state="active", server="spb", summary="Узел не отвечает"):
|
||||||
|
return {"fingerprint": fp, "status": {"state": state},
|
||||||
|
"labels": {"alertname": name, "severity": "critical", "server": server},
|
||||||
|
"annotations": {"summary": summary}}
|
||||||
|
|
||||||
|
|
||||||
|
def run(coro):
|
||||||
|
return asyncio.run(coro)
|
||||||
|
|
||||||
|
|
||||||
|
def test_new_alert_is_sent_once_to_subscribers(sent):
|
||||||
|
run(handlers.process_alerts([alert("a1")]))
|
||||||
|
run(handlers.process_alerts([alert("a1")]))
|
||||||
|
assert [uid for uid, _ in sent] == [1]
|
||||||
|
assert "Проблема · NodeDown" in sent[0][1]
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolved_alert_is_announced(sent):
|
||||||
|
run(handlers.process_alerts([alert("a1")]))
|
||||||
|
run(handlers.process_alerts([]))
|
||||||
|
assert "В норме · NodeDown" in sent[-1][1]
|
||||||
|
assert handlers.db.list_firing_fingerprints() == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_suppressed_alerts_are_ignored(sent):
|
||||||
|
run(handlers.process_alerts([alert("a1", state="suppressed"), {"status": {"state": "active"}}]))
|
||||||
|
assert sent == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_alert_text_is_html_escaped():
|
||||||
|
text = handlers._fmt_firing(alert("a1", name="<b>x</b>", summary="a < b & c"))
|
||||||
|
assert "<b>x</b>" in text
|
||||||
|
assert "a < b & c" in text
|
||||||
|
|
||||||
|
|
||||||
|
def test_watchdog_reports_once_and_recovers(sent):
|
||||||
|
wd = handlers.Watchdog(threshold=3)
|
||||||
|
for _ in range(5):
|
||||||
|
run(wd.failed("timeout"))
|
||||||
|
assert len(sent) == 1 and "Мониторинг недоступен" in sent[0][1]
|
||||||
|
run(wd.ok())
|
||||||
|
assert "Мониторинг снова доступен" in sent[-1][1]
|
||||||
|
run(wd.ok())
|
||||||
|
assert len(sent) == 2
|
||||||
|
|
||||||
|
|
||||||
|
def test_watchdog_ignores_short_glitches(sent):
|
||||||
|
wd = handlers.Watchdog(threshold=3)
|
||||||
|
run(wd.failed("timeout"))
|
||||||
|
run(wd.failed("timeout"))
|
||||||
|
run(wd.ok())
|
||||||
|
assert sent == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_token_is_sent_in_header_not_url():
|
||||||
|
assert handlers._auth("secret") == {"Authorization": "Bearer secret"}
|
||||||
69
tests/test_reports.py
Normal file
69
tests/test_reports.py
Normal file
|
|
@ -0,0 +1,69 @@
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
import handlers
|
||||||
|
from tests.conftest import vec
|
||||||
|
|
||||||
|
|
||||||
|
def run(coro):
|
||||||
|
return asyncio.run(coro)
|
||||||
|
|
||||||
|
|
||||||
|
def test_uptime_format():
|
||||||
|
assert handlers.fmt_uptime(3 * 86400 + 5 * 3600) == "3д 5ч"
|
||||||
|
assert handlers.fmt_uptime(7200) == "2ч"
|
||||||
|
assert handlers.fmt_uptime(-5) == "0ч"
|
||||||
|
|
||||||
|
|
||||||
|
def test_servers_report_marks_problems(prom):
|
||||||
|
import time
|
||||||
|
|
||||||
|
now = time.time()
|
||||||
|
prom.update({
|
||||||
|
"up{": vec(spb=1, ams=0),
|
||||||
|
"node_cpu_seconds_total": vec(spb=12, ams=3),
|
||||||
|
"MemAvailable": vec(spb=95, ams=40),
|
||||||
|
"node_filesystem_avail_bytes": vec(spb=50, ams=91),
|
||||||
|
"node_load1": vec(spb=0.5, ams=0.1),
|
||||||
|
"node_boot_time_seconds": vec(spb=now - 2 * 86400, ams=now - 3600),
|
||||||
|
"SwapTotal": vec(spb=0, ams=0),
|
||||||
|
"hwmon": vec(spb=40),
|
||||||
|
})
|
||||||
|
text = run(handlers.render_servers())
|
||||||
|
assert "<b>SPB</b> · ⚠ память" in text
|
||||||
|
assert "<b>AMS</b> · НЕ ОТВЕЧАЕТ" in text
|
||||||
|
assert "<b>DORM</b> — нет данных" in text
|
||||||
|
assert "аптайм 2д 0ч" in text
|
||||||
|
|
||||||
|
|
||||||
|
def test_services_report(prom):
|
||||||
|
prom["up"] = [
|
||||||
|
{"metric": {"job": "node_spb", "instance": "spb:9100"}, "value": [0, "1"]},
|
||||||
|
{"metric": {"job": "blackbox_http", "instance": "https://<bad>.example"}, "value": [0, "0"]},
|
||||||
|
]
|
||||||
|
text = run(handlers.render_services())
|
||||||
|
assert "1/2 в норме" in text
|
||||||
|
assert "<bad>" in text
|
||||||
|
|
||||||
|
|
||||||
|
def test_services_all_ok(prom):
|
||||||
|
prom["up"] = [{"metric": {"job": "node_spb"}, "value": [0, "1"]}]
|
||||||
|
assert "Все 1 в норме" in run(handlers.render_services())
|
||||||
|
|
||||||
|
|
||||||
|
def test_certs_report_sorted_with_warning(prom):
|
||||||
|
prom["probe_ssl_earliest_cert_expiry"] = [
|
||||||
|
{"metric": {"instance": "https://deev.space/"}, "value": [0, "60.2"]},
|
||||||
|
{"metric": {"instance": "https://tablo.deev.su"}, "value": [0, "5.1"]},
|
||||||
|
]
|
||||||
|
text = run(handlers.render_certs())
|
||||||
|
assert "скоро истекают: tablo.deev.su — 5 дн" in text
|
||||||
|
rows = text.split("<pre>")[1].split("</pre>")[0].splitlines()
|
||||||
|
assert rows[0].endswith("tablo.deev.su") and rows[1].endswith("deev.space")
|
||||||
|
|
||||||
|
|
||||||
|
def test_prometheus_error_is_reported(monkeypatch):
|
||||||
|
async def boom(session, query):
|
||||||
|
raise RuntimeError("connection refused")
|
||||||
|
|
||||||
|
monkeypatch.setattr(handlers, "_promq", boom)
|
||||||
|
assert "Не удалось получить метрики" in run(handlers.render_servers())
|
||||||
Loading…
Add table
Reference in a new issue