chore: auto-commit from dashboard
This commit is contained in:
@@ -9,6 +9,8 @@ from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
import requests
|
||||
|
||||
from src.config import Config
|
||||
from src.fast_commands import dispatch as fast_dispatch, set_channel_context
|
||||
from src.last_response_store import set_last as _set_last_response
|
||||
@@ -87,6 +89,63 @@ def _get_config() -> Config:
|
||||
return _config
|
||||
|
||||
|
||||
# Claude CLI rate-limit errors look like:
|
||||
# "Claude CLI error (exit 1): You've hit your session limit · resets 10:50am (UTC)"
|
||||
# (also seen for the total subscription limit, not just per-session — same phrasing).
|
||||
_RATE_LIMIT_RE = re.compile(r"hit your .*limit", re.IGNORECASE)
|
||||
|
||||
_LOCAL_FALLBACK_SYSTEM_PROMPT = (
|
||||
"Ești Echo, asistentul personal al lui Marius, dar rulezi temporar pe un "
|
||||
"model local mic pentru că Claude a atins limita de rate. Nu ai acces la "
|
||||
"unelte, memorie sau istoricul conversației — răspunde scurt și direct, "
|
||||
"doar la mesajul curent, în limba în care a fost scris."
|
||||
)
|
||||
|
||||
_LOCAL_FALLBACK_PREFIX = (
|
||||
"⚠️ Claude e la limită — răspund temporar pe un model local, mai simplu "
|
||||
"(fără istoric, fără unelte):\n\n"
|
||||
)
|
||||
|
||||
|
||||
def _is_rate_limit_error(err: Exception) -> bool:
|
||||
return bool(_RATE_LIMIT_RE.search(str(err)))
|
||||
|
||||
|
||||
def _local_fallback_reply(text: str) -> str | None:
|
||||
"""Best-effort reply from the local llama.cpp fallback (LXC 104, Qwen3.5-2B).
|
||||
|
||||
Returns None if the fallback itself is unreachable/fails, so the caller
|
||||
can fall back further to surfacing the original Claude error.
|
||||
"""
|
||||
cfg = _get_config().get("local_fallback", {}) or {}
|
||||
if not cfg.get("enabled", False):
|
||||
return None
|
||||
url = cfg.get("url")
|
||||
if not url:
|
||||
return None
|
||||
try:
|
||||
resp = requests.post(
|
||||
url,
|
||||
json={
|
||||
"messages": [
|
||||
{"role": "system", "content": _LOCAL_FALLBACK_SYSTEM_PROMPT},
|
||||
{"role": "user", "content": text},
|
||||
],
|
||||
"temperature": 0.3,
|
||||
"max_tokens": 500,
|
||||
},
|
||||
timeout=45,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
content = resp.json()["choices"][0]["message"]["content"].strip()
|
||||
if not content:
|
||||
return None
|
||||
return _LOCAL_FALLBACK_PREFIX + content
|
||||
except Exception as e: # noqa: BLE001
|
||||
log.error("Local fallback LLM failed: %s", e)
|
||||
return None
|
||||
|
||||
|
||||
def route_message(
|
||||
channel_id: str,
|
||||
user_id: str,
|
||||
@@ -218,6 +277,12 @@ def route_message(
|
||||
return response, False
|
||||
except Exception as e:
|
||||
log.error("Claude error for channel %s: %s", channel_id, e)
|
||||
if _is_rate_limit_error(e):
|
||||
log.warning("Rate limit detected for channel %s — trying local fallback", channel_id)
|
||||
fallback = _local_fallback_reply(text)
|
||||
if fallback is not None:
|
||||
_set_last_response(channel_id, fallback)
|
||||
return fallback, False
|
||||
return f"Error: {e}", False
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user