[eric] pro: a pool with nothing to serve with ends the turn with a card that names OpenSwarm Pro and the way around it, instead of parking the chat 21 minutes as a transient; the probe reports it too (ENG-462)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
ciregenz
2026-09-07 10:23:59 -07:00
co-authored by Claude Fable 5.1
parent 87ebcc47d7
commit 5d5a58fc29
5 changed files with 118 additions and 1 deletions
+4
View File
@@ -768,6 +768,10 @@ async def probe_model(body: dict):
except Exception as e:
msg = str(e).splitlines()[0] if str(e) else type(e).__name__
low = msg.lower()
from backend.apps.agents.core.error_classify import is_pool_outage
if is_pool_outage(e):
# The Pro pool with nothing to serve with is not a transient: every chat on it fails until a person re-authorises the accounts.
return {"ok": False, "error": "OpenSwarm Pro has no capacity right now: its shared accounts need re-authorising by the team. Your own key or a connected subscription will work."}
# Suppress transients: chat retries naturally and probe-time alias 404s often differ from chat resolution.
if any(s in low for s in (
"timeout", "timed out",
@@ -566,6 +566,16 @@ def is_transient_capacity_error(exc: BaseException, extra_text: str = "") -> boo
# Exponential-ish backoff schedule (seconds) for silently retrying a transient upstream capacity error before giving up and surfacing the rate-limit pill.
CAPACITY_BACKOFFS = [5, 15, 45, 90, 180]
# The OpenSwarm Pro proxy's own words when its shared pool has nothing to serve with (proxyForward.ts):
# both accounts auth_failed reads exactly like a busy second, and a busy second heals in 5 s while a
# dead pool needs a person. Past the silent backoffs the difference is the whole card.
P_POOL_OUTAGE = re.compile(r"no\s+pool\s+capacity|no\s+backup\s+available|primary\s+account\s+failed\s+auth", re.IGNORECASE)
@typechecked
def is_pool_outage(exc: BaseException, extra_text: str = "") -> bool:
return bool(P_POOL_OUTAGE.search(f"{exc!s}\n{extra_text}"))
@typechecked
def capacity_retry_wait(exc: BaseException, attempt: int, extra_text: str = "") -> Optional[int]:
@@ -29,6 +29,7 @@ from backend.apps.agents.core.error_classify import (
process_exit_code,
is_stale_tool_schema_error,
is_transient_capacity_error,
is_pool_outage,
is_free_trial_exhausted,
is_out_of_tokens,
has_auth_status,
@@ -396,6 +397,40 @@ async def handle_run_error(e: Exception, session: AgentSession, session_id: str,
})
except Exception:
logger.debug("submit_diagnostic cli_binary_missing failed", exc_info=True)
elif is_pool_outage(e, extra_text=p_stderr_tail):
# OpenSwarm Pro's shared pool answered "no capacity" through every silent backoff (335 s). Both
# of its accounts dead reads the same as one busy second, and the reconnect ladder would park
# this chat for 21 more minutes on a lane only a person can fix. Say so, once, and stop.
session.status = "completed" if turn.current_turn_emitted else "error"
friendly_msg = (
"OpenSwarm Pro has no capacity right now: its shared accounts are signed out and need "
"re-authorising by the team. Your own API key or a connected subscription will work in "
"the meantime (Settings > Models); send your message again once you have switched."
)
error_msg = Message(role="system", content=friendly_msg, branch_id=session.active_branch_id)
absorb_repeat_card(session, error_msg)
try:
from backend.apps.service.client import submit_diagnostic
submit_diagnostic({
"kind": "model_error",
"subkind": "pro_unavailable",
"model": session.model,
"provider": session.provider,
"error_preview": redact_for_telemetry(str(e), limit=400),
"flight": flight_recorder.build_envelope(session_id, "model_error", "openswarm_pro_unavailable", session.model, "stream" if turn.current_turn_emitted else "spawn", -1),
})
except Exception:
logger.debug("submit_diagnostic pro_unavailable failed", exc_info=True)
await ws_manager.send_to_session(session_id, "agent:auth_error", {
"session_id": session_id,
"reason": "openswarm_pro_unavailable",
"message": friendly_msg,
"model": session.model,
})
await ws_manager.send_to_session(session_id, "agent:message", {
"session_id": session_id,
"message": error_msg.model_dump(mode="json"),
})
elif is_transient_capacity_error(e, extra_text=p_stderr_tail):
# A genuine throttle (429/overload/capacity) that already burned the whole silent-backoff budget (the only way one reaches here). It's a limit, not a failure, so don't append a system-message card; emit a transient signal for the muted pill and mark the turn completed so it doesn't read as an error.
# 335s of ladder is a blip's worth of patience, and a closed lid or switched network outlasts it, so park and retry before conceding a turn the user never chose to end.
@@ -64,6 +64,14 @@ POLICY = "policy"
UNKNOWN = "unknown"
P_POOL_OUTAGE_WORDS = ("no pool capacity", "no backup available", "primary account failed auth")
def is_pool_outage_text(text: str) -> bool:
low = (text or "").lower()
return any(w in low for w in P_POOL_OUTAGE_WORDS)
class ProviderError(BaseModel):
"""What the provider said, normalised. `kind` drives the caller's choice of recovery."""
@@ -184,7 +192,9 @@ def user_facing_sentence(err: ProviderError, model: str) -> str:
has to whip an answer out of the agent, so a message that only diagnoses is a half-fix.
"""
who = "This model"
if err.lane in ("antigravity", "gc", "ag"):
if is_pool_outage_text(err.raw):
who = "OpenSwarm Pro"
elif err.lane in ("antigravity", "gc", "ag"):
who = "Gemini"
elif err.lane in ("codex", "cx"):
who = "ChatGPT"
+58
View File
@@ -0,0 +1,58 @@
"""OpenSwarm Pro's shared pool with nothing to serve with is a person's problem, not a transient: the chat says so once and stops."""
import inspect
import pytest
from backend.apps.agents.core.error_classify import is_pool_outage, is_transient_capacity_error
from backend.apps.agents.core.models import AgentSession
P_POOL_503 = 'API Error: 503 {"error":{"message":"No pool capacity available. Try again shortly."}}'
P_BACKUP_503 = "API Error: 503 Primary account failed auth, no backup available."
def test_the_proxy_words_are_recognised_and_stay_transient_for_the_silent_backoffs():
for text in (P_POOL_503, P_BACKUP_503):
assert is_pool_outage(RuntimeError(text))
assert is_transient_capacity_error(RuntimeError(text)), "the first 335 s of silent retries still run; a busy second heals there"
assert not is_pool_outage(RuntimeError("API Error: 503 Service Unavailable"))
assert not is_pool_outage(RuntimeError("overloaded_error"))
@pytest.mark.asyncio
async def test_past_the_backoffs_the_turn_ends_with_the_pro_card_and_never_parks(monkeypatch):
from backend.apps.agents.manager.run import handle_run_error as mod
from backend.apps.agents.manager.streaming.state import TurnState
sent = []
async def p_send(session_id, event, payload):
sent.append((event, payload))
monkeypatch.setattr(mod.ws_manager, "send_to_session", p_send)
envelopes = []
from backend.apps.service import client as p_client
monkeypatch.setattr(p_client, "submit_diagnostic", lambda d: envelopes.append(d))
s = AgentSession(name="t", model="sonnet-5", dashboard_id="d")
await mod.handle_run_error(RuntimeError(P_POOL_503), s, s.id, TurnState(), [])
events = [e for e, _ in sent]
assert "agent:reconnect_wait" not in events, "a dead pool must not park the chat on the 60/300/900 s ladder"
assert "agent:auth_error" in events and "agent:message" in events
reason = next(p for e, p in sent if e == "agent:auth_error")["reason"]
assert reason == "openswarm_pro_unavailable"
cards = [m for m in s.messages if m.role == "system"]
assert len(cards) == 1 and "OpenSwarm Pro" in cards[0].content and "own API key" in cards[0].content
assert s.status == "error"
assert [d.get("subkind") for d in envelopes] == ["pro_unavailable"]
def test_the_pool_branch_sits_above_the_generic_capacity_branch():
from backend.apps.agents.manager.run import handle_run_error as mod
src = inspect.getsource(mod.handle_run_error)
assert src.index("is_pool_outage(") < src.index("is_transient_capacity_error(e, extra_text=p_stderr_tail)"), "a veto goes at the top of the decision"
def test_the_sentence_names_openswarm_pro():
from backend.apps.agents.manager.streaming.provider_error_speech import ProviderError, user_facing_sentence
err = ProviderError(kind="overloaded", subscription_spent=False, status=503, lane=None, reset_seconds=None, raw=P_POOL_503)
assert user_facing_sentence(err, "sonnet-5").startswith("OpenSwarm Pro")