mirror of
https://github.com/EDeev/alertbot.git
synced 2026-10-07 20:49:56 +03:00
Токены в заголовке, сторож недоступности мониторинга, путь к базе из окружения
- токены Prometheus/Alertmanager передаются в заголовке Authorization, а не в адресе; - если Alertmanager не отвечает ALERT_WATCHDOG_FAILURES опросов подряд, бот сообщает «Мониторинг недоступен» и отдельно — когда он снова доступен; - один цикл опроса вынесен в process_alerts(); путь к SQLite — ALERTBOT_DB; - переводы строк приведены к LF.
This commit is contained in:
parent
1a77cf8620
commit
63c9976888
6 changed files with 556 additions and 503 deletions
|
|
@ -4,7 +4,8 @@ BOT_TOKEN=123456:your-telegram-bot-token
|
||||||
# Comma-separated Telegram user ids allowed to use the bot
|
# Comma-separated Telegram user ids allowed to use the bot
|
||||||
ALLOWED_IDS=111111111,222222222
|
ALLOWED_IDS=111111111,222222222
|
||||||
|
|
||||||
# Prometheus/Alertmanager endpoints (behind your own token-auth reverse-proxy path)
|
# Prometheus/Alertmanager endpoints behind your own reverse proxy.
|
||||||
|
# The bot sends "Authorization: Bearer <token>" — the proxy must check it.
|
||||||
AM_ALERTS_URL=https://your-domain/ambot/api/v2/alerts
|
AM_ALERTS_URL=https://your-domain/ambot/api/v2/alerts
|
||||||
PROM_QUERY_URL=https://your-domain/prombot/api/v1/query
|
PROM_QUERY_URL=https://your-domain/prombot/api/v1/query
|
||||||
AM_TOKEN=change-me
|
AM_TOKEN=change-me
|
||||||
|
|
@ -15,3 +16,8 @@ SERVERS_ORDER=srv1,srv2,srv3
|
||||||
|
|
||||||
ALERT_POLL_SECONDS=45
|
ALERT_POLL_SECONDS=45
|
||||||
ALERT_HTTP_TIMEOUT=15
|
ALERT_HTTP_TIMEOUT=15
|
||||||
|
# Report "monitoring unreachable" after this many failed polls in a row
|
||||||
|
ALERT_WATCHDOG_FAILURES=4
|
||||||
|
|
||||||
|
# SQLite file with subscriptions and alert state
|
||||||
|
ALERTBOT_DB=notifications.db
|
||||||
|
|
|
||||||
1
.gitattributes
vendored
Normal file
1
.gitattributes
vendored
Normal file
|
|
@ -0,0 +1 @@
|
||||||
|
* text=auto eol=lf
|
||||||
89
handlers.py
89
handlers.py
|
|
@ -10,12 +10,12 @@ from aiogram import BaseMiddleware, Router
|
||||||
from aiogram.filters import Command, CommandObject
|
from aiogram.filters import Command, CommandObject
|
||||||
from aiogram.types import Message
|
from aiogram.types import Message
|
||||||
|
|
||||||
from init import (ALERT_HTTP_TIMEOUT, ALERT_POLL_SECONDS, ALLOWED_IDS, AM_ALERTS_URL,
|
from init import (ALERT_HTTP_TIMEOUT, ALERT_POLL_SECONDS, ALERT_WATCHDOG_FAILURES, ALLOWED_IDS,
|
||||||
AM_TOKEN, PROM_QUERY_URL, PROM_TOKEN, bot)
|
AM_ALERTS_URL, AM_TOKEN, DB_PATH, PROM_QUERY_URL, PROM_TOKEN, bot)
|
||||||
from sql import DatabaseManager
|
from sql import DatabaseManager
|
||||||
|
|
||||||
router = Router()
|
router = Router()
|
||||||
db = DatabaseManager()
|
db = DatabaseManager(DB_PATH)
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
# Node names as used in Prometheus' "server" label — match your own scrape config.
|
# Node names as used in Prometheus' "server" label — match your own scrape config.
|
||||||
|
|
@ -128,8 +128,14 @@ async def cmd_certs(msg: Message):
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
# Prometheus
|
# Prometheus
|
||||||
# ======================================================================
|
# ======================================================================
|
||||||
|
def _auth(token: str) -> dict:
|
||||||
|
"""Token goes in the Authorization header, not in the URL (URLs end up in proxy logs)."""
|
||||||
|
return {"Authorization": f"Bearer {token}"}
|
||||||
|
|
||||||
|
|
||||||
async def _promq(session: aiohttp.ClientSession, query: str) -> List[dict]:
|
async def _promq(session: aiohttp.ClientSession, query: str) -> List[dict]:
|
||||||
async with session.get(PROM_QUERY_URL, params={"query": query, "token": PROM_TOKEN},
|
async with session.get(PROM_QUERY_URL, params={"query": query},
|
||||||
|
headers=_auth(PROM_TOKEN),
|
||||||
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
j = await r.json()
|
j = await r.json()
|
||||||
|
|
@ -277,8 +283,8 @@ def _fmt_resolved(name: str) -> str:
|
||||||
|
|
||||||
|
|
||||||
async def _fetch_alerts(session: aiohttp.ClientSession) -> List[dict]:
|
async def _fetch_alerts(session: aiohttp.ClientSession) -> List[dict]:
|
||||||
params = {"token": AM_TOKEN, "active": "true", "silenced": "false", "inhibited": "false"}
|
params = {"active": "true", "silenced": "false", "inhibited": "false"}
|
||||||
async with session.get(AM_ALERTS_URL, params=params,
|
async with session.get(AM_ALERTS_URL, params=params, headers=_auth(AM_TOKEN),
|
||||||
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
timeout=aiohttp.ClientTimeout(total=ALERT_HTTP_TIMEOUT)) as r:
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
return await r.json()
|
return await r.json()
|
||||||
|
|
@ -304,30 +310,67 @@ async def render_active_alerts() -> str:
|
||||||
return f"<b>Активные алерты: {len(firing)}</b>\n\n" + "\n\n".join(_fmt_firing(a) for a in firing)
|
return f"<b>Активные алерты: {len(firing)}</b>\n\n" + "\n\n".join(_fmt_firing(a) for a in firing)
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_watchdog_down(error: str) -> str:
|
||||||
|
return ("<b>Мониторинг недоступен</b>\nAlertmanager не отвечает, о новых проблемах бот сейчас не узнает.\n"
|
||||||
|
f"Последняя ошибка: {html.escape(error)}")
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_watchdog_up() -> str:
|
||||||
|
return "<b>Мониторинг снова доступен</b>"
|
||||||
|
|
||||||
|
|
||||||
|
async def process_alerts(data: List[dict]) -> None:
|
||||||
|
"""One poll: announce new firing alerts and alerts that have resolved since the last poll."""
|
||||||
|
current = {}
|
||||||
|
for a in data:
|
||||||
|
if a.get("status", {}).get("state") != "active":
|
||||||
|
continue
|
||||||
|
fp = a.get("fingerprint")
|
||||||
|
if not fp:
|
||||||
|
continue
|
||||||
|
current[fp] = a
|
||||||
|
if db.get_alert_status(fp) != "firing":
|
||||||
|
await _broadcast(_fmt_firing(a))
|
||||||
|
db.upsert_alert(fp, "firing", a.get("labels", {}).get("alertname", "alert"))
|
||||||
|
for fp, name in db.list_firing_fingerprints():
|
||||||
|
if fp not in current:
|
||||||
|
await _broadcast(_fmt_resolved(name))
|
||||||
|
db.upsert_alert(fp, "resolved", name)
|
||||||
|
db.purge_old_resolved()
|
||||||
|
|
||||||
|
|
||||||
|
class Watchdog:
|
||||||
|
"""Counts failed polls in a row; reports once when monitoring is lost and once when it is back."""
|
||||||
|
|
||||||
|
def __init__(self, threshold: int):
|
||||||
|
self.threshold = threshold
|
||||||
|
self.failures = 0
|
||||||
|
self.reported = False
|
||||||
|
|
||||||
|
async def failed(self, error: str) -> None:
|
||||||
|
self.failures += 1
|
||||||
|
if self.failures >= self.threshold and not self.reported:
|
||||||
|
self.reported = True
|
||||||
|
await _broadcast(_fmt_watchdog_down(error))
|
||||||
|
|
||||||
|
async def ok(self) -> None:
|
||||||
|
if self.reported:
|
||||||
|
await _broadcast(_fmt_watchdog_up())
|
||||||
|
self.failures = 0
|
||||||
|
self.reported = False
|
||||||
|
|
||||||
|
|
||||||
async def alert_poller():
|
async def alert_poller():
|
||||||
log.info("alert poller started (every %ss)", ALERT_POLL_SECONDS)
|
log.info("alert poller started (every %ss)", ALERT_POLL_SECONDS)
|
||||||
|
watchdog = Watchdog(ALERT_WATCHDOG_FAILURES)
|
||||||
async with aiohttp.ClientSession() as session:
|
async with aiohttp.ClientSession() as session:
|
||||||
while True:
|
while True:
|
||||||
try:
|
try:
|
||||||
data = await _fetch_alerts(session)
|
await process_alerts(await _fetch_alerts(session))
|
||||||
current = {}
|
await watchdog.ok()
|
||||||
for a in data:
|
|
||||||
if a.get("status", {}).get("state") != "active":
|
|
||||||
continue
|
|
||||||
fp = a.get("fingerprint")
|
|
||||||
if not fp:
|
|
||||||
continue
|
|
||||||
current[fp] = a
|
|
||||||
if db.get_alert_status(fp) != "firing":
|
|
||||||
await _broadcast(_fmt_firing(a))
|
|
||||||
db.upsert_alert(fp, "firing", a.get("labels", {}).get("alertname", "alert"))
|
|
||||||
for fp, name in db.list_firing_fingerprints():
|
|
||||||
if fp not in current:
|
|
||||||
await _broadcast(_fmt_resolved(name))
|
|
||||||
db.upsert_alert(fp, "resolved", name)
|
|
||||||
db.purge_old_resolved()
|
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
break
|
break
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
log.warning("alert poll failed: %s", e)
|
log.warning("alert poll failed: %s", e)
|
||||||
|
await watchdog.failed(str(e) or type(e).__name__)
|
||||||
await asyncio.sleep(ALERT_POLL_SECONDS)
|
await asyncio.sleep(ALERT_POLL_SECONDS)
|
||||||
|
|
|
||||||
3
init.py
3
init.py
|
|
@ -20,6 +20,9 @@ AM_TOKEN = os.environ["AM_TOKEN"]
|
||||||
PROM_TOKEN = os.environ["PROM_TOKEN"]
|
PROM_TOKEN = os.environ["PROM_TOKEN"]
|
||||||
ALERT_POLL_SECONDS = int(os.environ.get("ALERT_POLL_SECONDS", "45"))
|
ALERT_POLL_SECONDS = int(os.environ.get("ALERT_POLL_SECONDS", "45"))
|
||||||
ALERT_HTTP_TIMEOUT = int(os.environ.get("ALERT_HTTP_TIMEOUT", "15"))
|
ALERT_HTTP_TIMEOUT = int(os.environ.get("ALERT_HTTP_TIMEOUT", "15"))
|
||||||
|
# After this many failed polls in a row the bot reports that monitoring itself is unreachable
|
||||||
|
ALERT_WATCHDOG_FAILURES = int(os.environ.get("ALERT_WATCHDOG_FAILURES", "4"))
|
||||||
|
DB_PATH = os.environ.get("ALERTBOT_DB", "notifications.db")
|
||||||
|
|
||||||
bot = Bot(token=BOT_TOKEN, default=DefaultBotProperties(parse_mode=ParseMode.HTML))
|
bot = Bot(token=BOT_TOKEN, default=DefaultBotProperties(parse_mode=ParseMode.HTML))
|
||||||
dp = Dispatcher(storage=MemoryStorage())
|
dp = Dispatcher(storage=MemoryStorage())
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue