From bbf086329f7f70763d98112b3a459581570be5ae Mon Sep 17 00:00:00 2001 From: ciregenz Date: Fri, 7 Aug 2026 06:23:45 -0700 Subject: [PATCH] [eric] agents: a router-down turn names its cause instead of dying unclassified, found by the forced-failure battery --- .../core/is_router_unavailable_error.py | 27 +++++++++ .../agents/manager/configure_provider_env.py | 2 + .../agents/manager/run/handle_run_error.py | 60 ++++++++++--------- .../tests/test_router_unavailable_envelope.py | 42 +++++++++++++ 4 files changed, 103 insertions(+), 28 deletions(-) create mode 100644 backend/apps/agents/core/is_router_unavailable_error.py create mode 100644 backend/tests/test_router_unavailable_envelope.py diff --git a/backend/apps/agents/core/is_router_unavailable_error.py b/backend/apps/agents/core/is_router_unavailable_error.py new file mode 100644 index 00000000..cc3be237 --- /dev/null +++ b/backend/apps/agents/core/is_router_unavailable_error.py @@ -0,0 +1,27 @@ +"""Terminal shape of "our own router is down", kept beside the classifier it delegates to.""" + +import re + +from typeguard import typechecked + +from backend.apps.agents.core.error_classify import is_router_unreachable_error + + +@typechecked +def is_router_unavailable_error(text: str) -> bool: + """True when the turn died because our own localhost router was down, either because the CLI + could not reach it or because we refused to start at all. Distinct from + `is_router_unreachable_error`, which is the narrower mid-turn "resume and carry on" case: this + one is the terminal shape, and it exists so the envelope names a cause instead of shrugging + 'unclassified' at the one failure whose fix is entirely ours.""" + if not text.strip(): + return False + if is_router_unreachable_error(text): + return True + return bool(re.search( + r"9router\s+is\s+not\s+running" + r"|9router\s+could\s+not\s+start" + r"|9router.{0,30}not\s+ready", + text, + re.IGNORECASE, + )) diff --git a/backend/apps/agents/manager/configure_provider_env.py b/backend/apps/agents/manager/configure_provider_env.py index a7fa44f5..028263c5 100644 --- a/backend/apps/agents/manager/configure_provider_env.py +++ b/backend/apps/agents/manager/configure_provider_env.py @@ -232,6 +232,8 @@ async def configure_provider_env( logger.info(f"[MCP-DEBUG] Using 9Router (api_type={api_type})") else: # router_available() above already attempted a revival; reaching here means it truly can't start. + from backend.apps.agents.core import flight_recorder + flight_recorder.crumb(session.id, "router-unavailable", model=session.model, api=api_type) if api_type != "anthropic" or resolved_is_9router: raise ValueError( f"9Router is not running; cannot use {session.model}. " diff --git a/backend/apps/agents/manager/run/handle_run_error.py b/backend/apps/agents/manager/run/handle_run_error.py index 0f7cf3cf..a1f61d01 100644 --- a/backend/apps/agents/manager/run/handle_run_error.py +++ b/backend/apps/agents/manager/run/handle_run_error.py @@ -22,6 +22,7 @@ from backend.apps.agents.core.error_classify import ( is_unknown_model_error, parse_retry_after, ) +from backend.apps.agents.core.is_router_unavailable_error import is_router_unavailable_error from backend.apps.agents.core.extract_reset_hint import extract_reset_hint from backend.apps.agents.core.redact_for_telemetry import redact_for_telemetry from backend.apps.agents.core import flight_recorder @@ -29,6 +30,25 @@ from backend.apps.agents.core import flight_recorder logger = logging.getLogger(__name__) +@typechecked +def p_report_model_error(subkind: str, session_id: str, session: AgentSession, turn: TurnState, + e: BaseException, stderr_tail: str) -> None: + """The three terminal model_error rungs differ only by subkind, so they share one submitter.""" + try: + from backend.apps.service.client import submit_diagnostic + submit_diagnostic({ + "kind": "model_error", + "subkind": subkind, + "flight": flight_recorder.build_envelope(session_id, "model_error", subkind, session.model, "stream" if turn.current_turn_emitted else "spawn", -1), + "model": session.model, + "provider": session.provider, + "connection_mode": getattr(load_settings(), "connection_mode", "own_key"), + "error_preview": redact_for_telemetry(str(e), limit=400), + "stderr_tail": redact_for_telemetry(stderr_tail), + }) + except Exception: + logger.debug(f"submit_diagnostic {subkind} failed", exc_info=True) + @typechecked async def handle_run_error(e: Exception, session: AgentSession, session_id: str, turn: TurnState, p_stderr_buffer: List[str]) -> None: logger.exception(f"Agent {session_id} error: {e}") @@ -246,20 +266,17 @@ async def handle_run_error(e: Exception, session: AgentSession, session_id: str, }) elif is_unknown_model_error(e, extra_text=p_stderr_tail): # Upstream rejected the model code itself (e.g. Codex 1211 on a ChatGPT plan that lacks our GPT ids). Track it; the friendly "add an API key / pick another model" card is rendered frontend-side. - try: - from backend.apps.service.client import submit_diagnostic - submit_diagnostic({ - "kind": "model_error", - "subkind": "unknown_model", - "flight": flight_recorder.build_envelope(session_id, "model_error", "unknown_model", session.model, "stream" if turn.current_turn_emitted else "spawn", -1), - "model": session.model, - "provider": session.provider, - "connection_mode": getattr(load_settings(), "connection_mode", "own_key"), - "error_preview": redact_for_telemetry(str(e), limit=400), - "stderr_tail": redact_for_telemetry(p_stderr_tail), - }) - except Exception: - logger.debug("submit_diagnostic model_error failed", exc_info=True) + p_report_model_error("unknown_model", session_id, session, turn, e, p_stderr_tail) + error_msg = Message(role="system", content=f"Error: {str(e)}", branch_id=session.active_branch_id) + session.messages.append(error_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": error_msg.model_dump(mode="json"), + }) + elif is_router_unavailable_error(f"{e} {p_stderr_tail}"): + # Our own router is down. Naming it beats "unclassified": this is the one failure family + # where the fix is entirely on our side of the wire. + p_report_model_error("router_unavailable", session_id, session, turn, e, p_stderr_tail) error_msg = Message(role="system", content=f"Error: {str(e)}", branch_id=session.active_branch_id) session.messages.append(error_msg) await ws_manager.send_to_session(session_id, "agent:message", { @@ -268,20 +285,7 @@ async def handle_run_error(e: Exception, session: AgentSession, session_id: str, }) else: # Track unclassified agent failures too so we stop flying blind on them. - try: - from backend.apps.service.client import submit_diagnostic - submit_diagnostic({ - "kind": "model_error", - "subkind": "unclassified", - "flight": flight_recorder.build_envelope(session_id, "model_error", "unclassified", session.model, "stream" if turn.current_turn_emitted else "spawn", -1), - "model": session.model, - "provider": session.provider, - "connection_mode": getattr(load_settings(), "connection_mode", "own_key"), - "error_preview": redact_for_telemetry(str(e), limit=400), - "stderr_tail": redact_for_telemetry(p_stderr_tail), - }) - except Exception: - logger.debug("submit_diagnostic model_error failed", exc_info=True) + p_report_model_error("unclassified", session_id, session, turn, e, p_stderr_tail) # The SDK's ProcessError masks the cause behind "Check stderr output for details"; append the scrubbed stderr tail so the card (and its analytics copy) names what actually broke instead of shipping a dead end. p_card_text = f"Error: {str(e)}" p_cause = redact_for_telemetry(p_stderr_tail, limit=400).strip() diff --git a/backend/tests/test_router_unavailable_envelope.py b/backend/tests/test_router_unavailable_envelope.py new file mode 100644 index 00000000..3f84a3c8 --- /dev/null +++ b/backend/tests/test_router_unavailable_envelope.py @@ -0,0 +1,42 @@ +"""The router-down envelope must NAME the cause. + +Found by the forced-failure battery on 2026-08-07: holding port 20128 with a dead socket so 9Router +could not rebind produced a real terminal failure whose envelope read `subkind=unclassified`, with a +breadcrumb trail that simply stopped after the prep phases. The cause was sitting in plain text in +`error_preview` ("9Router is not running; cannot use sonnet-cc") but nothing could be queried on it. +""" + +import inspect + +from backend.apps.agents.core.error_classify import is_router_unreachable_error +from backend.apps.agents.core.is_router_unavailable_error import is_router_unavailable_error +from backend.apps.agents.manager.run import handle_run_error + + +def test_the_verbatim_live_refusal_is_classified(): + # Exact string raised by configure_provider_env and captured in the battery envelope. + assert is_router_unavailable_error( + "9Router is not running; cannot use sonnet-cc. Install Node.js and restart the app, " + "or switch to a model with a direct API key." + ) + + +def test_the_mid_turn_unreachable_shapes_still_qualify(): + for text in ("API Error: Unable to connect. Is the computer able to access the url?", + "fetch failed", "connect ECONNREFUSED 127.0.0.1:20128"): + assert is_router_unavailable_error(text), text + assert is_router_unreachable_error(text), "the narrower resume-path check must keep matching too" + + +def test_unrelated_failures_are_not_swallowed(): + for text in ("", " ", "Prompt is too long", "Invalid API key", + "The router of the story is that nothing broke"): + assert not is_router_unavailable_error(text), text + + +def test_the_rung_sits_above_unclassified(): + src = inspect.getsource(handle_run_error) + assert "is_router_unavailable_error" in src + assert src.index("is_router_unavailable_error") < src.index('p_report_model_error("unclassified"'), \ + "a router death must be named before the catch-all claims it" + assert 'p_report_model_error("router_unavailable"' in src