feat(steering): mesaje mid-tur + /stop pe turul în zbor

Un al doilea mesaj trimis cât Claude încă lucra aștepta până se termina
turul 1 — corecția „stai, nu în master" ajungea după ce greșeala era gata.
Verificat în producție înainte de commit: mesajul 2 stătea 25s blocat în
lock, apoi pornea ca tur separat.

Acum canalele de chat pot ține un proces `claude` viu per canal, cu stdin
deschis, și al doilea mesaj intră în ACELAȘI tur.

- `src/claude_runner.py` — ClaudeProcess (steering, respawn cu --resume,
  drenare stderr, respawn la comutarea OpenRouter) + RunnerRegistry
  (max_live, reaper pe inactivitate, stop_all la shutdown)
- `src/stream_json.py` — parser stream-json partajat cu `_run_claude`;
  pur, nu aruncă niciodată pe is_error (PlanningSession retrimite pe
  error_max_turns și depinde de asta)
- `src/sentinels.py` — un singur loc pentru __AUDIO__/__STEERED__, în loc
  de 4 verificări copiate; repară și bug-ul preexistent prin care
  WhatsApp posta literal `__AUDIO__:/cale`
- dispecer în `send_message`: lock.acquire(blocking=False) — eșecul de a
  lua lock-ul ESTE „rulează un tur", ceea ce elimină flagul inflight din
  decizie și cursa TOCTOU odată cu el
- `/stop` oprește turul, nu sesiunea — active.json rămâne valid
- rate limit prin proces persistent vine ca result.is_error, nu ca exit
  code; convertit înapoi în același RuntimeError, altfel fallback-ul
  local nu s-ar mai declanșa niciodată, în tăcere

Steering-ul nu face niciodată cross-adapter (un mesaj text nu intră
într-un tur voice: împart același channel_id). Mesajele steered dintr-un
tur care pică sunt re-livrate, nu pierdute.

Testat live cu CLI-ul real: corecție la secunda 10 dintr-un tur de 24s,
un singur result, num_turns=2. Notă: mesajele steered sunt împachetate în
[EXTERNAL CONTENT], deci o corecție formulată ca override agresiv poate
fi refuzată ca prompt injection — pentru oprire folosește /stop.

Suită: 1199 passed, 12 failed (toate pre-existente pe HEAD curat).

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SiJGsZVSEGjRHZEJiXaxCC
This commit is contained in:
2026-09-02 11:05:58 +00:00
parent 2d8ee9b581
commit 747afbaf9d
23 changed files with 4183 additions and 95 deletions

View File

@@ -1,15 +1,18 @@
"""Tests for the local LLM fallback: history, tools, net status, web search."""
import time
from pathlib import Path
from unittest.mock import MagicMock, patch
import pytest
from src import fallback_history, net_status, web_search
from src import claude_runner, claude_session, fallback_history, net_status, router, web_search
from src import local_fallback_tools as lft
from src.router import (_forced_tool, _is_creative_request, _is_text_task,
_parse_tool_args, _run_fallback_tools)
FAKE_CLAUDE = Path(__file__).parent / "fake_claude.py"
@pytest.fixture(autouse=True)
def clean_history():
@@ -623,3 +626,217 @@ class TestFallbackNeverDropsTurn:
assert "răspuns direct" in reply
conv.assert_called_once()
# ---------------------------------------------------------------------------
# T6 — persistent-process rate limit must still reach the fallback
# ---------------------------------------------------------------------------
#
# The one-shot `claude -p` dies with a nonzero exit code at the limit; a
# persistent steering process (src/claude_runner.py) reports the same limit
# as `result.is_error` in the stream instead — it never exits. `ClaudeProcess`
# converts that back into the exact `RuntimeError` the one-shot path raises
# (covered in tests/test_claude_runner.py). What isn't covered anywhere else
# is the far end of the wire: that error genuinely reaching router.py's
# handler and firing `_local_fallback_reply` for real. Untested, this is how
# the single most-worked-on subsystem in the repo goes dead in silence
# (tasks/steering-plan.md, blockers C3/C4, task T6 — "the 2am Friday test").
def _local_fallback_cfg():
"""`_get_config()` double: only `local_fallback` is stubbed — every
other key falls through to its caller-supplied default, same as the
real `Config().get(key, default)`. A blanket `return_value` (as used
for the router-only tests above) would also hijack route_message's own
`_get_config().get("bot.default_model", "sonnet")` lookup and feed a
dict into `--model`."""
cfg = MagicMock()
cfg.get.side_effect = lambda key, default=None: (
{"enabled": True, "url": "http://x"} if key == "local_fallback" else default
)
return cfg
@pytest.fixture(autouse=True)
def _reset_steering_registry():
"""H2: the steering registry is module-level global state — never let
a live (fake) process from one test leak into the next."""
claude_runner.reset_registry_for_tests()
yield
claude_runner.reset_registry_for_tests()
@pytest.fixture
def temp_sessions(tmp_path, monkeypatch):
"""Isolated sessions/active.json so these tests never touch the real one."""
sessions_dir = tmp_path / "sessions"
sessions_dir.mkdir()
sf = sessions_dir / "active.json"
sf.write_text("{}")
monkeypatch.setattr(claude_session, "SESSIONS_DIR", sessions_dir)
monkeypatch.setattr(claude_session, "_SESSIONS_FILE", sf)
return sf
@pytest.fixture
def steering_on(monkeypatch, temp_sessions):
"""Steering enabled, wired to the real tests/fake_claude.py subprocess —
a genuine persistent process, not a mock, so the rate-limit path fires
the way it actually does in production."""
monkeypatch.setattr(claude_runner, "CLAUDE_BIN", str(FAKE_CLAUDE))
monkeypatch.setattr(claude_session, "_steering_config", lambda *a, **kw: (True, 2, 20))
class TestPersistentRateLimitReachesFallback:
"""T6 (`pytest -k persistent`): a rate limit arriving as `result.is_error`
from a PERSISTENT process must still invoke the local model — asserting
the RESCUE happened, not merely that some exception was raised. A test
that only checks "an exception was raised" is exactly the bug this task
exists to catch: that's also true the day the fallback silently stops
firing."""
def test_persistent_rate_limit_invokes_real_fallback(self, steering_on, monkeypatch):
monkeypatch.setenv("FAKE_CLAUDE_SCENARIO", "rate_limit")
with patch("src.router._get_config", return_value=_local_fallback_cfg()), \
patch("src.router.set_channel_context"), \
patch("src.router._call_local_llm", return_value={"content": "raspuns local"}), \
patch("src.router._local_fallback_reply",
wraps=router._local_fallback_reply) as fallback_spy:
result, is_cmd = router.route_message("ch-t6-persist", "user-1", "salut")
fallback_spy.assert_called_once_with("salut", channel_id="ch-t6-persist")
assert "raspuns local" in result
assert is_cmd is False
class _RateLimitAfterSteerProc:
"""Minimal ClaudeProcess double: the turn raises the rate-limit
RuntimeError while leaving one already-steered text behind in
`_pending_steers` — the shape of a real ClaudeProcess whose turn a
`steer()` landed in, then died (C3). `ClaudeProcess`'s own
thread-timing for how a steer lands mid-turn is exercised for real in
tests/test_claude_runner.py; this double exists only to drive
router.py's redelivery path deterministically."""
def __init__(self, channel_id, model=None, session_id=None, cwd=None):
self.channel_id = channel_id
self.session_id = session_id or "fake-sid"
self.inflight = False
self._pending_steers = ["steered while you were away"]
def run_turn(self, text, on_text=None, timeout=300):
self.inflight = True
raise RuntimeError(
"Claude CLI error (exit 1): You've hit your session limit · resets 10am (UTC)"
)
def pop_pending_steers(self):
pending, self._pending_steers = self._pending_steers, []
return pending
class _SingleProcRegistry:
"""RunnerRegistry double that always hands back the one proc it was
built with."""
def __init__(self, proc):
self._procs = {proc.channel_id: proc}
def get(self, channel_id, model=None, session_id=None, cwd=None):
return self._procs.get(channel_id)
def stop(self, channel_id):
return self._procs.pop(channel_id, None) is not None
def _echo_llm(url, messages, tools=None, temperature=0.0):
"""`_call_local_llm` double whose reply names the user text it saw, so
the original turn's reply and the steered turn's reply stay
distinguishable however many passes `_local_fallback_reply` makes."""
return {"content": f"echo:{messages[-1]['content']}"}
class TestSteeredTextDeliveredOnRateLimit:
"""C3: the plan is explicit that T6 must test DELIVERY, not just
detection. A message steered into a turn that then dies on a rate
limit must still produce its own answer to the user — its request
thread already returned `__STEERED__` and is gone, so `on_text` is the
only channel left."""
def test_steered_reply_delivered_via_on_text(self, monkeypatch, temp_sessions):
proc = _RateLimitAfterSteerProc("ch-t6-c3")
registry = _SingleProcRegistry(proc)
# Both names: `_dispatch_steering` calls `get_registry()`, but
# `claude_session.pop_pending_steers` (C3's redelivery hook) peeks
# the module-level `_registry` directly rather than calling
# `get_registry()` again (that would spin one up as a side effect
# for a channel that never used steering).
monkeypatch.setattr(claude_runner, "get_registry", lambda **kw: registry)
monkeypatch.setattr(claude_runner, "_registry", registry)
monkeypatch.setattr(claude_session, "_steering_config", lambda *a, **kw: (True, 2, 20))
streamed = []
with patch("src.router._get_config", return_value=_local_fallback_cfg()), \
patch("src.router.set_channel_context"), \
patch("src.router._call_local_llm", side_effect=_echo_llm):
result, is_cmd = router.route_message(
"ch-t6-c3", "user-1", "mesaj original", on_text=streamed.append,
)
# The original message's own answer still comes back as the
# function's return value.
assert "echo:mesaj original" in result
assert is_cmd is False
# The steered message never had a request thread of its own left to
# return to — its answer must have gone out through on_text.
assert len(streamed) == 1
assert "echo:steered while you were away" in streamed[0]
class TestNonRateLimitErrorStaysQuiet:
"""C4 regression guard: verified in tasks/steering-plan.md that
`PlanningSession` retries on `error_max_turns`, which depends on the
runner RETURNING (never raising) a non-rate-limit `is_error`. If a
future change 'simplifies' that into a raise, this test catches it by
failing on the wrong side: the fallback would fire when it must not."""
def test_generic_is_error_does_not_raise_or_call_fallback(self, steering_on, monkeypatch):
monkeypatch.setenv("FAKE_CLAUDE_SCENARIO", "generic_error")
with patch("src.router._local_fallback_reply") as fallback:
result, is_cmd = router.route_message("ch-t6-c4", "user-1", "salut")
fallback.assert_not_called()
assert "hit max turns" in result
assert is_cmd is False
class TestSteeringOffRollback:
"""The flag is the rollback mechanism (steering-plan.md Etapa 2) — it
has to actually roll the rate-limit -> fallback path back to today's
behavior, not merely skip steering-specific code paths."""
def test_steering_off_still_reaches_fallback_via_one_shot_path(
self, monkeypatch, temp_sessions,
):
monkeypatch.setattr(claude_session, "_steering_config", lambda *a, **kw: (False, 0, 20))
monkeypatch.setattr(
claude_session, "_run_claude",
lambda *a, **kw: (_ for _ in ()).throw(RuntimeError(
"Claude CLI error (exit 1): You've hit your session limit · resets 10am (UTC)"
)),
)
with patch("src.router._get_config", return_value=_local_fallback_cfg()), \
patch("src.router.set_channel_context"), \
patch("src.router._call_local_llm", return_value={"content": "raspuns local"}), \
patch("src.router._local_fallback_reply",
wraps=router._local_fallback_reply) as fallback_spy:
result, is_cmd = router.route_message("ch-t6-off", "user-1", "salut")
fallback_spy.assert_called_once_with("salut", channel_id="ch-t6-off")
assert "raspuns local" in result
assert is_cmd is False
# Steering played no role at all — the registry was never even created.
assert claude_runner._registry is None