feat(steering): mesaje mid-tur + /stop pe turul în zbor
Un al doilea mesaj trimis cât Claude încă lucra aștepta până se termina turul 1 — corecția „stai, nu în master" ajungea după ce greșeala era gata. Verificat în producție înainte de commit: mesajul 2 stătea 25s blocat în lock, apoi pornea ca tur separat. Acum canalele de chat pot ține un proces `claude` viu per canal, cu stdin deschis, și al doilea mesaj intră în ACELAȘI tur. - `src/claude_runner.py` — ClaudeProcess (steering, respawn cu --resume, drenare stderr, respawn la comutarea OpenRouter) + RunnerRegistry (max_live, reaper pe inactivitate, stop_all la shutdown) - `src/stream_json.py` — parser stream-json partajat cu `_run_claude`; pur, nu aruncă niciodată pe is_error (PlanningSession retrimite pe error_max_turns și depinde de asta) - `src/sentinels.py` — un singur loc pentru __AUDIO__/__STEERED__, în loc de 4 verificări copiate; repară și bug-ul preexistent prin care WhatsApp posta literal `__AUDIO__:/cale` - dispecer în `send_message`: lock.acquire(blocking=False) — eșecul de a lua lock-ul ESTE „rulează un tur", ceea ce elimină flagul inflight din decizie și cursa TOCTOU odată cu el - `/stop` oprește turul, nu sesiunea — active.json rămâne valid - rate limit prin proces persistent vine ca result.is_error, nu ca exit code; convertit înapoi în același RuntimeError, altfel fallback-ul local nu s-ar mai declanșa niciodată, în tăcere Steering-ul nu face niciodată cross-adapter (un mesaj text nu intră într-un tur voice: împart același channel_id). Mesajele steered dintr-un tur care pică sunt re-livrate, nu pierdute. Testat live cu CLI-ul real: corecție la secunda 10 dintr-un tur de 24s, un singur result, num_turns=2. Notă: mesajele steered sunt împachetate în [EXTERNAL CONTENT], deci o corecție formulată ca override agresiv poate fi refuzată ca prompt injection — pentru oprire folosește /stop. Suită: 1199 passed, 12 failed (toate pre-existente pe HEAD curat). Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SiJGsZVSEGjRHZEJiXaxCC
This commit is contained in:
124
tests/fake_claude.py
Executable file
124
tests/fake_claude.py
Executable file
@@ -0,0 +1,124 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fake `claude` CLI binary for tests/test_claude_runner.py.
|
||||
|
||||
Runs as a REAL subprocess (spawned by `subprocess.Popen`, exactly like the
|
||||
real CLI) so steering genuinely exercises stdin/stdout pipe timing that a
|
||||
mock `Popen` can't reproduce. Speaks the subset of stream-json that
|
||||
`src/claude_runner.py` depends on:
|
||||
|
||||
- one `system`/`init` event with a `session_id`, on the first turn only
|
||||
- one `assistant` event per response, with a text block
|
||||
- one `result` event per turn
|
||||
|
||||
Controlled entirely via environment variables (this process has no access
|
||||
to the test's Python objects):
|
||||
|
||||
FAKE_CLAUDE_SCENARIO normal | steer | timeout | rate_limit |
|
||||
big_stderr | crash_immediately (default: normal)
|
||||
FAKE_CLAUDE_SESSION_ID session_id to report (default: fake-session-1)
|
||||
|
||||
The `system`/`init` event also carries the process's own `argv` — tests use
|
||||
this to assert `--resume <sid>` was actually passed on respawn, without
|
||||
needing a separate side channel.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
def _emit(obj: dict) -> None:
|
||||
print(json.dumps(obj), flush=True)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
scenario = os.environ.get("FAKE_CLAUDE_SCENARIO", "normal")
|
||||
session_id = os.environ.get("FAKE_CLAUDE_SESSION_ID", "fake-session-1")
|
||||
|
||||
if scenario == "crash_immediately":
|
||||
sys.exit(1)
|
||||
|
||||
if scenario == "big_stderr":
|
||||
# T3: without a stderr-draining thread on the caller's side, this
|
||||
# fills the OS pipe buffer (~64KB) and blocks forever, before this
|
||||
# process ever gets to read a turn off stdin.
|
||||
for _ in range(4000):
|
||||
print("x" * 40, file=sys.stderr, flush=True)
|
||||
|
||||
first_turn = True
|
||||
for raw_line in sys.stdin:
|
||||
raw_line = raw_line.strip()
|
||||
if not raw_line:
|
||||
continue
|
||||
try:
|
||||
msg = json.loads(raw_line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
text = msg.get("message", {}).get("content", "")
|
||||
|
||||
if first_turn:
|
||||
_emit({
|
||||
"type": "system", "subtype": "init",
|
||||
"session_id": session_id, "argv": sys.argv,
|
||||
})
|
||||
first_turn = False
|
||||
|
||||
if scenario == "timeout":
|
||||
time.sleep(600) # never responds — caller's own watchdog must kill us
|
||||
return
|
||||
|
||||
if scenario == "rate_limit":
|
||||
_emit({
|
||||
"type": "result", "subtype": "error", "session_id": session_id,
|
||||
"result": "You've hit your session limit · resets 10:50am (UTC)",
|
||||
"is_error": True,
|
||||
})
|
||||
continue
|
||||
|
||||
if scenario == "generic_error":
|
||||
# is_error=True but NOT a rate limit — C4: must be returned,
|
||||
# never raised, so a future PlanningSession-style caller can
|
||||
# still retry on `subtype` instead of catching an exception.
|
||||
_emit({
|
||||
"type": "result", "subtype": "error_max_turns", "session_id": session_id,
|
||||
"result": "hit max turns for this task", "is_error": True,
|
||||
})
|
||||
continue
|
||||
|
||||
if scenario == "steer":
|
||||
# Simulate a turn in progress that then reads a SECOND stdin
|
||||
# line (the steer) before finishing — exactly what a real
|
||||
# steering exchange looks like (spike: one `result`, num_turns=2).
|
||||
_emit({"type": "assistant", "message": {"content": [
|
||||
{"type": "text", "text": f"got:{text}"},
|
||||
]}})
|
||||
second_raw = sys.stdin.readline().strip()
|
||||
if second_raw:
|
||||
try:
|
||||
second_msg = json.loads(second_raw)
|
||||
second_text = second_msg.get("message", {}).get("content", "")
|
||||
except json.JSONDecodeError:
|
||||
second_text = ""
|
||||
_emit({"type": "assistant", "message": {"content": [
|
||||
{"type": "text", "text": f"steered:{second_text}"},
|
||||
]}})
|
||||
_emit({
|
||||
"type": "result", "subtype": "success", "session_id": session_id,
|
||||
"result": "done", "is_error": False, "num_turns": 2,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
})
|
||||
continue
|
||||
|
||||
# normal
|
||||
_emit({"type": "assistant", "message": {"content": [
|
||||
{"type": "text", "text": f"echo:{text}"},
|
||||
]}})
|
||||
_emit({
|
||||
"type": "result", "subtype": "success", "session_id": session_id,
|
||||
"result": f"echo:{text}", "is_error": False, "num_turns": 1,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user