From 5d5a58fc292b3649b7ef9046e00feca3983d4808 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Mon, 7 Sep 2026 10:23:59 -0700 Subject: [PATCH] [eric] pro: a pool with nothing to serve with ends the turn with a card that names OpenSwarm Pro and the way around it, instead of parking the chat 21 minutes as a transient; the probe reports it too (ENG-462) Co-Authored-By: Claude Fable 5.1 --- backend/apps/agents/agents.py | 4 ++ backend/apps/agents/core/error_classify.py | 10 ++++ .../agents/manager/run/handle_run_error.py | 35 +++++++++++ .../streaming/provider_error_speech.py | 12 +++- backend/tests/test_pro_pool_outage.py | 58 +++++++++++++++++++ 5 files changed, 118 insertions(+), 1 deletion(-) create mode 100644 backend/tests/test_pro_pool_outage.py diff --git a/backend/apps/agents/agents.py b/backend/apps/agents/agents.py index b4664e12..1ccf3a59 100644 --- a/backend/apps/agents/agents.py +++ b/backend/apps/agents/agents.py @@ -768,6 +768,10 @@ async def probe_model(body: dict): except Exception as e: msg = str(e).splitlines()[0] if str(e) else type(e).__name__ low = msg.lower() + from backend.apps.agents.core.error_classify import is_pool_outage + if is_pool_outage(e): + # The Pro pool with nothing to serve with is not a transient: every chat on it fails until a person re-authorises the accounts. + return {"ok": False, "error": "OpenSwarm Pro has no capacity right now: its shared accounts need re-authorising by the team. Your own key or a connected subscription will work."} # Suppress transients: chat retries naturally and probe-time alias 404s often differ from chat resolution. if any(s in low for s in ( "timeout", "timed out", diff --git a/backend/apps/agents/core/error_classify.py b/backend/apps/agents/core/error_classify.py index 34bf9acf..49216e4d 100644 --- a/backend/apps/agents/core/error_classify.py +++ b/backend/apps/agents/core/error_classify.py @@ -566,6 +566,16 @@ def is_transient_capacity_error(exc: BaseException, extra_text: str = "") -> boo # Exponential-ish backoff schedule (seconds) for silently retrying a transient upstream capacity error before giving up and surfacing the rate-limit pill. CAPACITY_BACKOFFS = [5, 15, 45, 90, 180] +# The OpenSwarm Pro proxy's own words when its shared pool has nothing to serve with (proxyForward.ts): +# both accounts auth_failed reads exactly like a busy second, and a busy second heals in 5 s while a +# dead pool needs a person. Past the silent backoffs the difference is the whole card. +P_POOL_OUTAGE = re.compile(r"no\s+pool\s+capacity|no\s+backup\s+available|primary\s+account\s+failed\s+auth", re.IGNORECASE) + + +@typechecked +def is_pool_outage(exc: BaseException, extra_text: str = "") -> bool: + return bool(P_POOL_OUTAGE.search(f"{exc!s}\n{extra_text}")) + @typechecked def capacity_retry_wait(exc: BaseException, attempt: int, extra_text: str = "") -> Optional[int]: diff --git a/backend/apps/agents/manager/run/handle_run_error.py b/backend/apps/agents/manager/run/handle_run_error.py index e4cb8169..e722e9c1 100644 --- a/backend/apps/agents/manager/run/handle_run_error.py +++ b/backend/apps/agents/manager/run/handle_run_error.py @@ -29,6 +29,7 @@ from backend.apps.agents.core.error_classify import ( process_exit_code, is_stale_tool_schema_error, is_transient_capacity_error, + is_pool_outage, is_free_trial_exhausted, is_out_of_tokens, has_auth_status, @@ -396,6 +397,40 @@ async def handle_run_error(e: Exception, session: AgentSession, session_id: str, }) except Exception: logger.debug("submit_diagnostic cli_binary_missing failed", exc_info=True) + elif is_pool_outage(e, extra_text=p_stderr_tail): + # OpenSwarm Pro's shared pool answered "no capacity" through every silent backoff (335 s). Both + # of its accounts dead reads the same as one busy second, and the reconnect ladder would park + # this chat for 21 more minutes on a lane only a person can fix. Say so, once, and stop. + session.status = "completed" if turn.current_turn_emitted else "error" + friendly_msg = ( + "OpenSwarm Pro has no capacity right now: its shared accounts are signed out and need " + "re-authorising by the team. Your own API key or a connected subscription will work in " + "the meantime (Settings > Models); send your message again once you have switched." + ) + error_msg = Message(role="system", content=friendly_msg, branch_id=session.active_branch_id) + absorb_repeat_card(session, error_msg) + try: + from backend.apps.service.client import submit_diagnostic + submit_diagnostic({ + "kind": "model_error", + "subkind": "pro_unavailable", + "model": session.model, + "provider": session.provider, + "error_preview": redact_for_telemetry(str(e), limit=400), + "flight": flight_recorder.build_envelope(session_id, "model_error", "openswarm_pro_unavailable", session.model, "stream" if turn.current_turn_emitted else "spawn", -1), + }) + except Exception: + logger.debug("submit_diagnostic pro_unavailable failed", exc_info=True) + await ws_manager.send_to_session(session_id, "agent:auth_error", { + "session_id": session_id, + "reason": "openswarm_pro_unavailable", + "message": friendly_msg, + "model": session.model, + }) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": error_msg.model_dump(mode="json"), + }) elif is_transient_capacity_error(e, extra_text=p_stderr_tail): # A genuine throttle (429/overload/capacity) that already burned the whole silent-backoff budget (the only way one reaches here). It's a limit, not a failure, so don't append a system-message card; emit a transient signal for the muted pill and mark the turn completed so it doesn't read as an error. # 335s of ladder is a blip's worth of patience, and a closed lid or switched network outlasts it, so park and retry before conceding a turn the user never chose to end. diff --git a/backend/apps/agents/manager/streaming/provider_error_speech.py b/backend/apps/agents/manager/streaming/provider_error_speech.py index 18123ed8..ffe648e3 100644 --- a/backend/apps/agents/manager/streaming/provider_error_speech.py +++ b/backend/apps/agents/manager/streaming/provider_error_speech.py @@ -64,6 +64,14 @@ POLICY = "policy" UNKNOWN = "unknown" +P_POOL_OUTAGE_WORDS = ("no pool capacity", "no backup available", "primary account failed auth") + + +def is_pool_outage_text(text: str) -> bool: + low = (text or "").lower() + return any(w in low for w in P_POOL_OUTAGE_WORDS) + + class ProviderError(BaseModel): """What the provider said, normalised. `kind` drives the caller's choice of recovery.""" @@ -184,7 +192,9 @@ def user_facing_sentence(err: ProviderError, model: str) -> str: has to whip an answer out of the agent, so a message that only diagnoses is a half-fix. """ who = "This model" - if err.lane in ("antigravity", "gc", "ag"): + if is_pool_outage_text(err.raw): + who = "OpenSwarm Pro" + elif err.lane in ("antigravity", "gc", "ag"): who = "Gemini" elif err.lane in ("codex", "cx"): who = "ChatGPT" diff --git a/backend/tests/test_pro_pool_outage.py b/backend/tests/test_pro_pool_outage.py new file mode 100644 index 00000000..3d0935ea --- /dev/null +++ b/backend/tests/test_pro_pool_outage.py @@ -0,0 +1,58 @@ +"""OpenSwarm Pro's shared pool with nothing to serve with is a person's problem, not a transient: the chat says so once and stops.""" +import inspect + +import pytest + +from backend.apps.agents.core.error_classify import is_pool_outage, is_transient_capacity_error +from backend.apps.agents.core.models import AgentSession + +P_POOL_503 = 'API Error: 503 {"error":{"message":"No pool capacity available. Try again shortly."}}' +P_BACKUP_503 = "API Error: 503 Primary account failed auth, no backup available." + + +def test_the_proxy_words_are_recognised_and_stay_transient_for_the_silent_backoffs(): + for text in (P_POOL_503, P_BACKUP_503): + assert is_pool_outage(RuntimeError(text)) + assert is_transient_capacity_error(RuntimeError(text)), "the first 335 s of silent retries still run; a busy second heals there" + assert not is_pool_outage(RuntimeError("API Error: 503 Service Unavailable")) + assert not is_pool_outage(RuntimeError("overloaded_error")) + + +@pytest.mark.asyncio +async def test_past_the_backoffs_the_turn_ends_with_the_pro_card_and_never_parks(monkeypatch): + from backend.apps.agents.manager.run import handle_run_error as mod + from backend.apps.agents.manager.streaming.state import TurnState + + sent = [] + + async def p_send(session_id, event, payload): + sent.append((event, payload)) + + monkeypatch.setattr(mod.ws_manager, "send_to_session", p_send) + envelopes = [] + from backend.apps.service import client as p_client + monkeypatch.setattr(p_client, "submit_diagnostic", lambda d: envelopes.append(d)) + s = AgentSession(name="t", model="sonnet-5", dashboard_id="d") + await mod.handle_run_error(RuntimeError(P_POOL_503), s, s.id, TurnState(), []) + + events = [e for e, _ in sent] + assert "agent:reconnect_wait" not in events, "a dead pool must not park the chat on the 60/300/900 s ladder" + assert "agent:auth_error" in events and "agent:message" in events + reason = next(p for e, p in sent if e == "agent:auth_error")["reason"] + assert reason == "openswarm_pro_unavailable" + cards = [m for m in s.messages if m.role == "system"] + assert len(cards) == 1 and "OpenSwarm Pro" in cards[0].content and "own API key" in cards[0].content + assert s.status == "error" + assert [d.get("subkind") for d in envelopes] == ["pro_unavailable"] + + +def test_the_pool_branch_sits_above_the_generic_capacity_branch(): + from backend.apps.agents.manager.run import handle_run_error as mod + src = inspect.getsource(mod.handle_run_error) + assert src.index("is_pool_outage(") < src.index("is_transient_capacity_error(e, extra_text=p_stderr_tail)"), "a veto goes at the top of the decision" + + +def test_the_sentence_names_openswarm_pro(): + from backend.apps.agents.manager.streaming.provider_error_speech import ProviderError, user_facing_sentence + err = ProviderError(kind="overloaded", subscription_spent=False, status=503, lane=None, reset_seconds=None, raw=P_POOL_503) + assert user_facing_sentence(err, "sonnet-5").startswith("OpenSwarm Pro")