diff --git a/proxmox/lxc171-claude-agent/discord-bridge/README.md b/proxmox/lxc171-claude-agent/discord-bridge/README.md index 62cc1a4..ab4dc08 100644 --- a/proxmox/lxc171-claude-agent/discord-bridge/README.md +++ b/proxmox/lxc171-claude-agent/discord-bridge/README.md @@ -298,6 +298,7 @@ tail -2 ~/.claude-discord/logs/alerts.log | Simptom | Ce faci | |---------|---------| | Botul nu raspunde deloc in Discord | `systemctl --user status claude-discord`. Daca e `failed`, `journalctl --user -u claude-discord -n 100`. Cauza #1: token invalid sau **MESSAGE CONTENT INTENT** oprit. | +| Botul e viu dar tace, iar in log curg `WSServerHandshakeError: 503` | Gatewayul regional memorat ca `resume_gateway_url` a picat, iar discord.py il reincearca la infinit. Verifica: `curl -si --http1.1 -H 'Connection: Upgrade' -H 'Upgrade: websocket' -H 'Sec-WebSocket-Version: 13' -H 'Sec-WebSocket-Key: dGhlIHNhbXBsZSBub25jZQ==' 'https://gateway.discord.gg/?v=10&encoding=json'` — un `101` acolo plus `503` pe hostul regional confirma. Watchdogul iese singur dupa `GATEWAY_WATCHDOG_S` (implicit 300s) si systemd reporneste; daca vrei mai repede, `systemctl --user restart claude-discord`. | | Botul e viu dar ignora mesajele | Allowlist. Verifica `DISCORD_GUILD_IDS` / `DISCORD_CHANNEL_IDS` / `DISCORD_USER_IDS` din env. Respingerea e **tacuta**, intentionat. | | Comenzile `/` nu apar in lista din Discord | Botul a fost invitat fara scope-ul `applications.commands`. In log: `sync de comenzi slash esuat` + linkul de reinvitare. Reinvita botul, apoi `systemctl --user restart claude-discord`. | | Unitul se invarte in restart | Dupa 5 porniri esuate in 300s systemd renunta si lasa unitul `failed` (e voit). Repara, apoi `systemctl --user reset-failed claude-discord && systemctl --user start claude-discord`. | diff --git a/proxmox/lxc171-claude-agent/discord-bridge/bot.py b/proxmox/lxc171-claude-agent/discord-bridge/bot.py index eb395a9..b88f5d0 100644 --- a/proxmox/lxc171-claude-agent/discord-bridge/bot.py +++ b/proxmox/lxc171-claude-agent/discord-bridge/bot.py @@ -18,6 +18,7 @@ import io import logging import os import pathlib +import threading import time from dataclasses import dataclass, field @@ -1011,6 +1012,51 @@ class Bridge: await self.say(channel, text) +# ------------------------------------------------------------------ watchdog +# Iesire cu cod nenul => systemd (Restart=on-failure) reporneste procesul. +WATCHDOG_EXIT_CODE = 3 + + +class GatewayWatchdog: + """Masoara de cat timp e gatewayul cazut. + + discord.py memoreaza `resume_gateway_url` primit la ultimul READY si il + refoloseste la fiecare reconectare. Daca acel gateway regional pica + (503 la handshake pe gateway-us-east-1a), botul reincearca la infinit + acelasi host mort, cu backoff care creste pana la ~15 minute: procesul e + viu, deci systemd nu vede nimic si nici dashboardul. Singura iesire e un + IDENTIFY nou pe gateway.discord.gg, adica un restart. + """ + + def __init__(self, threshold_s: float, now=time.monotonic): + self.threshold_s = threshold_s + self._now = now + self.down_since: float | None = None + + def on_up(self) -> None: + self.down_since = None + + def on_down(self) -> None: + # Doar prima cadere conteaza: reincercarile care esueaza nu reseteaza ceasul. + if self.down_since is None: + self.down_since = self._now() + + def downtime(self) -> float: + return 0.0 if self.down_since is None else self._now() - self.down_since + + def expired(self) -> bool: + # threshold <= 0 dezactiveaza watchdogul. + return self.threshold_s > 0 and self.downtime() >= self.threshold_s + + +def _hard_exit_after(seconds: float, code: int) -> None: # pragma: no cover - plasa de siguranta + """Daca `close()` se blocheaza pe un socket mort, iesim oricum.""" + + timer = threading.Timer(seconds, lambda: os._exit(code)) + timer.daemon = True + timer.start() + + # --------------------------------------------------------------- client real def make_client(bridge: Bridge | None = None): # pragma: no cover - are nevoie de discord.py if discord is None: @@ -1028,13 +1074,53 @@ def make_client(bridge: Bridge | None = None): # pragma: no cover - are nevoie self.bridge.get_channel = self.get_channel self.tree = commands_slash.build_tree(self, self.bridge) self._started = False + self.watchdog = GatewayWatchdog(config.get_float("GATEWAY_WATCHDOG_S", 300.0)) + self.watchdog_tripped = False + self._watchdog_task = None async def setup_hook(self): # Sync PE GUILD: e instantaneu, spre deosebire de cel global (~1h). # Esecul nu doboara botul — mesajele obisnuite merg mai departe. await commands_slash.sync_guilds(self.tree, guild_ids()) + self._watchdog_task = asyncio.create_task(self._watchdog_loop()) + + async def _watchdog_loop(self): + # Verificam des in raport cu pragul, ca sa nu adaugam intarziere peste el. + interval = max(5.0, min(30.0, self.watchdog.threshold_s / 4)) + while not self.is_closed(): + await asyncio.sleep(interval) + if not self.watchdog.expired(): + continue + log.error( + "gateway cazut de %.0fs (prag %.0fs): ies cu codul %d ca systemd sa reporneasca", + self.watchdog.downtime(), + self.watchdog.threshold_s, + WATCHDOG_EXIT_CODE, + ) + with contextlib.suppress(Exception): + alerts.alert( + "WARN", + "punte Discord: gateway cazut, restart", + f"reconectarea a esuat {self.watchdog.downtime():.0f}s la rand; ies " + f"cu codul {WATCHDOG_EXIT_CODE} pentru un IDENTIFY nou.", + "gateway-watchdog", + ) + self.watchdog_tripped = True + _hard_exit_after(30.0, WATCHDOG_EXIT_CODE) + await self.close() + return + + async def on_connect(self): + self.watchdog.on_up() + + async def on_resumed(self): + self.watchdog.on_up() + + async def on_disconnect(self): + self.watchdog.on_down() async def on_ready(self): + self.watchdog.on_up() self.bridge.self_id = str(self.user.id) if self.user else None if not self._started: self._started = True @@ -1045,6 +1131,12 @@ def make_client(bridge: Bridge | None = None): # pragma: no cover - are nevoie await self.bridge.handle_message(message) async def close(self): + task = self._watchdog_task + self._watchdog_task = None + if task is not None and task is not asyncio.current_task(): + task.cancel() + with contextlib.suppress(asyncio.CancelledError): + await task await self.bridge.shutdown() await super().close() @@ -1079,7 +1171,8 @@ def main() -> int: # pragma: no cover return 2 client = make_client() client.run(token, log_handler=None) - return 0 + # Nenul: doar asa `Restart=on-failure` reporneste puntea. + return WATCHDOG_EXIT_CODE if getattr(client, "watchdog_tripped", False) else 0 if __name__ == "__main__": # pragma: no cover diff --git a/proxmox/lxc171-claude-agent/discord-bridge/tests/test_bot.py b/proxmox/lxc171-claude-agent/discord-bridge/tests/test_bot.py index 5754be6..47caaa2 100644 --- a/proxmox/lxc171-claude-agent/discord-bridge/tests/test_bot.py +++ b/proxmox/lxc171-claude-agent/discord-bridge/tests/test_bot.py @@ -421,3 +421,54 @@ async def test_cererea_de_aprobare_posteaza_butoane_in_fir(bridge, allowed): assert "rm -rf /tmp/x" in ch.texts[0] if bot.discord is not None: assert ch.sent[0].kwargs.get("view") is not None + + +# ---------------------------------------------------------------- watchdog + + +class FakeClock: + def __init__(self): + self.t = 0.0 + + def __call__(self): + return self.t + + +def test_watchdogul_tace_cat_timp_gatewayul_e_sus(): + clock = FakeClock() + wd = bot.GatewayWatchdog(300.0, now=clock) + wd.on_up() + clock.t = 10_000.0 # bot linistit ore intregi: niciun eveniment intre timp + assert wd.downtime() == 0.0 + assert not wd.expired() + + +def test_watchdogul_iarta_o_reconectare_scurta(): + clock = FakeClock() + wd = bot.GatewayWatchdog(300.0, now=clock) + wd.on_down() + clock.t = 4.0 + wd.on_up() # RESUMED + clock.t = 900.0 + assert not wd.expired() + + +def test_watchdogul_expira_dupa_prag_iar_reincercarile_nu_reseteaza_ceasul(): + clock = FakeClock() + wd = bot.GatewayWatchdog(300.0, now=clock) + wd.on_down() + for t in (60.0, 120.0, 240.0): # fiecare handshake 503 mai da un on_disconnect + clock.t = t + wd.on_down() + assert not wd.expired() + clock.t = 300.0 + assert wd.expired() + assert wd.downtime() == 300.0 + + +def test_watchdogul_se_poate_dezactiva(): + clock = FakeClock() + wd = bot.GatewayWatchdog(0.0, now=clock) + wd.on_down() + clock.t = 100_000.0 + assert not wd.expired()