fix(discord): watchdog de gateway — procesul viu pe un gateway mort nu mai trece neobservat
Puntea a tacut ~7 ore fara ca nimic sa semnaleze: discord.py memoreaza `resume_gateway_url` primit la ultimul READY si il refoloseste la fiecare reconectare. Cand gatewayul regional (gateway-us-east-1a) a inceput sa dea 503 la handshake, botul a reincercat la infinit acelasi host mort, cu backoff pana la ~15 minute. Procesul era viu, deci nici systemd nici dashboardul nu aveau ce vedea, iar `gateway.discord.gg` raspundea normal tot timpul. GatewayWatchdog numara de cat timp e legatura jos (on_connect / on_resumed / on_ready o ridica, on_disconnect o coboara, iar reincercarile esuate nu reseteaza ceasul). Peste `GATEWAY_WATCHDOG_S` (implicit 300s) alerteaza, iese cu codul 3 si lasa systemd sa reporneasca — restartul e singurul lucru care forteaza un IDENTIFY nou pe gateway.discord.gg. Un prag <= 0 il dezactiveaza. O iesire la 300s ramane sub StartLimitBurst=5/5min, deci bucla de restart nu poate ajunge sa lase unitul `failed`. `_hard_exit_after` acopera cazul in care `close()` se blocheaza pe socketul mort. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Q4uzvgm7AyJch5WH8QHRhY
This commit is contained in:
@@ -18,6 +18,7 @@ import io
|
||||
import logging
|
||||
import os
|
||||
import pathlib
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
@@ -1011,6 +1012,51 @@ class Bridge:
|
||||
await self.say(channel, text)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ watchdog
|
||||
# Iesire cu cod nenul => systemd (Restart=on-failure) reporneste procesul.
|
||||
WATCHDOG_EXIT_CODE = 3
|
||||
|
||||
|
||||
class GatewayWatchdog:
|
||||
"""Masoara de cat timp e gatewayul cazut.
|
||||
|
||||
discord.py memoreaza `resume_gateway_url` primit la ultimul READY si il
|
||||
refoloseste la fiecare reconectare. Daca acel gateway regional pica
|
||||
(503 la handshake pe gateway-us-east-1a), botul reincearca la infinit
|
||||
acelasi host mort, cu backoff care creste pana la ~15 minute: procesul e
|
||||
viu, deci systemd nu vede nimic si nici dashboardul. Singura iesire e un
|
||||
IDENTIFY nou pe gateway.discord.gg, adica un restart.
|
||||
"""
|
||||
|
||||
def __init__(self, threshold_s: float, now=time.monotonic):
|
||||
self.threshold_s = threshold_s
|
||||
self._now = now
|
||||
self.down_since: float | None = None
|
||||
|
||||
def on_up(self) -> None:
|
||||
self.down_since = None
|
||||
|
||||
def on_down(self) -> None:
|
||||
# Doar prima cadere conteaza: reincercarile care esueaza nu reseteaza ceasul.
|
||||
if self.down_since is None:
|
||||
self.down_since = self._now()
|
||||
|
||||
def downtime(self) -> float:
|
||||
return 0.0 if self.down_since is None else self._now() - self.down_since
|
||||
|
||||
def expired(self) -> bool:
|
||||
# threshold <= 0 dezactiveaza watchdogul.
|
||||
return self.threshold_s > 0 and self.downtime() >= self.threshold_s
|
||||
|
||||
|
||||
def _hard_exit_after(seconds: float, code: int) -> None: # pragma: no cover - plasa de siguranta
|
||||
"""Daca `close()` se blocheaza pe un socket mort, iesim oricum."""
|
||||
|
||||
timer = threading.Timer(seconds, lambda: os._exit(code))
|
||||
timer.daemon = True
|
||||
timer.start()
|
||||
|
||||
|
||||
# --------------------------------------------------------------- client real
|
||||
def make_client(bridge: Bridge | None = None): # pragma: no cover - are nevoie de discord.py
|
||||
if discord is None:
|
||||
@@ -1028,13 +1074,53 @@ def make_client(bridge: Bridge | None = None): # pragma: no cover - are nevoie
|
||||
self.bridge.get_channel = self.get_channel
|
||||
self.tree = commands_slash.build_tree(self, self.bridge)
|
||||
self._started = False
|
||||
self.watchdog = GatewayWatchdog(config.get_float("GATEWAY_WATCHDOG_S", 300.0))
|
||||
self.watchdog_tripped = False
|
||||
self._watchdog_task = None
|
||||
|
||||
async def setup_hook(self):
|
||||
# Sync PE GUILD: e instantaneu, spre deosebire de cel global (~1h).
|
||||
# Esecul nu doboara botul — mesajele obisnuite merg mai departe.
|
||||
await commands_slash.sync_guilds(self.tree, guild_ids())
|
||||
self._watchdog_task = asyncio.create_task(self._watchdog_loop())
|
||||
|
||||
async def _watchdog_loop(self):
|
||||
# Verificam des in raport cu pragul, ca sa nu adaugam intarziere peste el.
|
||||
interval = max(5.0, min(30.0, self.watchdog.threshold_s / 4))
|
||||
while not self.is_closed():
|
||||
await asyncio.sleep(interval)
|
||||
if not self.watchdog.expired():
|
||||
continue
|
||||
log.error(
|
||||
"gateway cazut de %.0fs (prag %.0fs): ies cu codul %d ca systemd sa reporneasca",
|
||||
self.watchdog.downtime(),
|
||||
self.watchdog.threshold_s,
|
||||
WATCHDOG_EXIT_CODE,
|
||||
)
|
||||
with contextlib.suppress(Exception):
|
||||
alerts.alert(
|
||||
"WARN",
|
||||
"punte Discord: gateway cazut, restart",
|
||||
f"reconectarea a esuat {self.watchdog.downtime():.0f}s la rand; ies "
|
||||
f"cu codul {WATCHDOG_EXIT_CODE} pentru un IDENTIFY nou.",
|
||||
"gateway-watchdog",
|
||||
)
|
||||
self.watchdog_tripped = True
|
||||
_hard_exit_after(30.0, WATCHDOG_EXIT_CODE)
|
||||
await self.close()
|
||||
return
|
||||
|
||||
async def on_connect(self):
|
||||
self.watchdog.on_up()
|
||||
|
||||
async def on_resumed(self):
|
||||
self.watchdog.on_up()
|
||||
|
||||
async def on_disconnect(self):
|
||||
self.watchdog.on_down()
|
||||
|
||||
async def on_ready(self):
|
||||
self.watchdog.on_up()
|
||||
self.bridge.self_id = str(self.user.id) if self.user else None
|
||||
if not self._started:
|
||||
self._started = True
|
||||
@@ -1045,6 +1131,12 @@ def make_client(bridge: Bridge | None = None): # pragma: no cover - are nevoie
|
||||
await self.bridge.handle_message(message)
|
||||
|
||||
async def close(self):
|
||||
task = self._watchdog_task
|
||||
self._watchdog_task = None
|
||||
if task is not None and task is not asyncio.current_task():
|
||||
task.cancel()
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
await task
|
||||
await self.bridge.shutdown()
|
||||
await super().close()
|
||||
|
||||
@@ -1079,7 +1171,8 @@ def main() -> int: # pragma: no cover
|
||||
return 2
|
||||
client = make_client()
|
||||
client.run(token, log_handler=None)
|
||||
return 0
|
||||
# Nenul: doar asa `Restart=on-failure` reporneste puntea.
|
||||
return WATCHDOG_EXIT_CODE if getattr(client, "watchdog_tripped", False) else 0
|
||||
|
||||
|
||||
if __name__ == "__main__": # pragma: no cover
|
||||
|
||||
Reference in New Issue
Block a user