mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-28 04:24:51 +02:00
[eric] pro: a pool with nothing to serve with ends the turn with a card that names OpenSwarm Pro and the way around it, instead of parking the chat 21 minutes as a transient; the probe reports it too (ENG-462)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5.1
parent
87ebcc47d7
commit
5d5a58fc29
@@ -768,6 +768,10 @@ async def probe_model(body: dict):
|
||||
except Exception as e:
|
||||
msg = str(e).splitlines()[0] if str(e) else type(e).__name__
|
||||
low = msg.lower()
|
||||
from backend.apps.agents.core.error_classify import is_pool_outage
|
||||
if is_pool_outage(e):
|
||||
# The Pro pool with nothing to serve with is not a transient: every chat on it fails until a person re-authorises the accounts.
|
||||
return {"ok": False, "error": "OpenSwarm Pro has no capacity right now: its shared accounts need re-authorising by the team. Your own key or a connected subscription will work."}
|
||||
# Suppress transients: chat retries naturally and probe-time alias 404s often differ from chat resolution.
|
||||
if any(s in low for s in (
|
||||
"timeout", "timed out",
|
||||
|
||||
@@ -566,6 +566,16 @@ def is_transient_capacity_error(exc: BaseException, extra_text: str = "") -> boo
|
||||
# Exponential-ish backoff schedule (seconds) for silently retrying a transient upstream capacity error before giving up and surfacing the rate-limit pill.
|
||||
CAPACITY_BACKOFFS = [5, 15, 45, 90, 180]
|
||||
|
||||
# The OpenSwarm Pro proxy's own words when its shared pool has nothing to serve with (proxyForward.ts):
|
||||
# both accounts auth_failed reads exactly like a busy second, and a busy second heals in 5 s while a
|
||||
# dead pool needs a person. Past the silent backoffs the difference is the whole card.
|
||||
P_POOL_OUTAGE = re.compile(r"no\s+pool\s+capacity|no\s+backup\s+available|primary\s+account\s+failed\s+auth", re.IGNORECASE)
|
||||
|
||||
|
||||
@typechecked
|
||||
def is_pool_outage(exc: BaseException, extra_text: str = "") -> bool:
|
||||
return bool(P_POOL_OUTAGE.search(f"{exc!s}\n{extra_text}"))
|
||||
|
||||
|
||||
@typechecked
|
||||
def capacity_retry_wait(exc: BaseException, attempt: int, extra_text: str = "") -> Optional[int]:
|
||||
|
||||
@@ -29,6 +29,7 @@ from backend.apps.agents.core.error_classify import (
|
||||
process_exit_code,
|
||||
is_stale_tool_schema_error,
|
||||
is_transient_capacity_error,
|
||||
is_pool_outage,
|
||||
is_free_trial_exhausted,
|
||||
is_out_of_tokens,
|
||||
has_auth_status,
|
||||
@@ -396,6 +397,40 @@ async def handle_run_error(e: Exception, session: AgentSession, session_id: str,
|
||||
})
|
||||
except Exception:
|
||||
logger.debug("submit_diagnostic cli_binary_missing failed", exc_info=True)
|
||||
elif is_pool_outage(e, extra_text=p_stderr_tail):
|
||||
# OpenSwarm Pro's shared pool answered "no capacity" through every silent backoff (335 s). Both
|
||||
# of its accounts dead reads the same as one busy second, and the reconnect ladder would park
|
||||
# this chat for 21 more minutes on a lane only a person can fix. Say so, once, and stop.
|
||||
session.status = "completed" if turn.current_turn_emitted else "error"
|
||||
friendly_msg = (
|
||||
"OpenSwarm Pro has no capacity right now: its shared accounts are signed out and need "
|
||||
"re-authorising by the team. Your own API key or a connected subscription will work in "
|
||||
"the meantime (Settings > Models); send your message again once you have switched."
|
||||
)
|
||||
error_msg = Message(role="system", content=friendly_msg, branch_id=session.active_branch_id)
|
||||
absorb_repeat_card(session, error_msg)
|
||||
try:
|
||||
from backend.apps.service.client import submit_diagnostic
|
||||
submit_diagnostic({
|
||||
"kind": "model_error",
|
||||
"subkind": "pro_unavailable",
|
||||
"model": session.model,
|
||||
"provider": session.provider,
|
||||
"error_preview": redact_for_telemetry(str(e), limit=400),
|
||||
"flight": flight_recorder.build_envelope(session_id, "model_error", "openswarm_pro_unavailable", session.model, "stream" if turn.current_turn_emitted else "spawn", -1),
|
||||
})
|
||||
except Exception:
|
||||
logger.debug("submit_diagnostic pro_unavailable failed", exc_info=True)
|
||||
await ws_manager.send_to_session(session_id, "agent:auth_error", {
|
||||
"session_id": session_id,
|
||||
"reason": "openswarm_pro_unavailable",
|
||||
"message": friendly_msg,
|
||||
"model": session.model,
|
||||
})
|
||||
await ws_manager.send_to_session(session_id, "agent:message", {
|
||||
"session_id": session_id,
|
||||
"message": error_msg.model_dump(mode="json"),
|
||||
})
|
||||
elif is_transient_capacity_error(e, extra_text=p_stderr_tail):
|
||||
# A genuine throttle (429/overload/capacity) that already burned the whole silent-backoff budget (the only way one reaches here). It's a limit, not a failure, so don't append a system-message card; emit a transient signal for the muted pill and mark the turn completed so it doesn't read as an error.
|
||||
# 335s of ladder is a blip's worth of patience, and a closed lid or switched network outlasts it, so park and retry before conceding a turn the user never chose to end.
|
||||
|
||||
@@ -64,6 +64,14 @@ POLICY = "policy"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
|
||||
P_POOL_OUTAGE_WORDS = ("no pool capacity", "no backup available", "primary account failed auth")
|
||||
|
||||
|
||||
def is_pool_outage_text(text: str) -> bool:
|
||||
low = (text or "").lower()
|
||||
return any(w in low for w in P_POOL_OUTAGE_WORDS)
|
||||
|
||||
|
||||
class ProviderError(BaseModel):
|
||||
"""What the provider said, normalised. `kind` drives the caller's choice of recovery."""
|
||||
|
||||
@@ -184,7 +192,9 @@ def user_facing_sentence(err: ProviderError, model: str) -> str:
|
||||
has to whip an answer out of the agent, so a message that only diagnoses is a half-fix.
|
||||
"""
|
||||
who = "This model"
|
||||
if err.lane in ("antigravity", "gc", "ag"):
|
||||
if is_pool_outage_text(err.raw):
|
||||
who = "OpenSwarm Pro"
|
||||
elif err.lane in ("antigravity", "gc", "ag"):
|
||||
who = "Gemini"
|
||||
elif err.lane in ("codex", "cx"):
|
||||
who = "ChatGPT"
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
"""OpenSwarm Pro's shared pool with nothing to serve with is a person's problem, not a transient: the chat says so once and stops."""
|
||||
import inspect
|
||||
|
||||
import pytest
|
||||
|
||||
from backend.apps.agents.core.error_classify import is_pool_outage, is_transient_capacity_error
|
||||
from backend.apps.agents.core.models import AgentSession
|
||||
|
||||
P_POOL_503 = 'API Error: 503 {"error":{"message":"No pool capacity available. Try again shortly."}}'
|
||||
P_BACKUP_503 = "API Error: 503 Primary account failed auth, no backup available."
|
||||
|
||||
|
||||
def test_the_proxy_words_are_recognised_and_stay_transient_for_the_silent_backoffs():
|
||||
for text in (P_POOL_503, P_BACKUP_503):
|
||||
assert is_pool_outage(RuntimeError(text))
|
||||
assert is_transient_capacity_error(RuntimeError(text)), "the first 335 s of silent retries still run; a busy second heals there"
|
||||
assert not is_pool_outage(RuntimeError("API Error: 503 Service Unavailable"))
|
||||
assert not is_pool_outage(RuntimeError("overloaded_error"))
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_past_the_backoffs_the_turn_ends_with_the_pro_card_and_never_parks(monkeypatch):
|
||||
from backend.apps.agents.manager.run import handle_run_error as mod
|
||||
from backend.apps.agents.manager.streaming.state import TurnState
|
||||
|
||||
sent = []
|
||||
|
||||
async def p_send(session_id, event, payload):
|
||||
sent.append((event, payload))
|
||||
|
||||
monkeypatch.setattr(mod.ws_manager, "send_to_session", p_send)
|
||||
envelopes = []
|
||||
from backend.apps.service import client as p_client
|
||||
monkeypatch.setattr(p_client, "submit_diagnostic", lambda d: envelopes.append(d))
|
||||
s = AgentSession(name="t", model="sonnet-5", dashboard_id="d")
|
||||
await mod.handle_run_error(RuntimeError(P_POOL_503), s, s.id, TurnState(), [])
|
||||
|
||||
events = [e for e, _ in sent]
|
||||
assert "agent:reconnect_wait" not in events, "a dead pool must not park the chat on the 60/300/900 s ladder"
|
||||
assert "agent:auth_error" in events and "agent:message" in events
|
||||
reason = next(p for e, p in sent if e == "agent:auth_error")["reason"]
|
||||
assert reason == "openswarm_pro_unavailable"
|
||||
cards = [m for m in s.messages if m.role == "system"]
|
||||
assert len(cards) == 1 and "OpenSwarm Pro" in cards[0].content and "own API key" in cards[0].content
|
||||
assert s.status == "error"
|
||||
assert [d.get("subkind") for d in envelopes] == ["pro_unavailable"]
|
||||
|
||||
|
||||
def test_the_pool_branch_sits_above_the_generic_capacity_branch():
|
||||
from backend.apps.agents.manager.run import handle_run_error as mod
|
||||
src = inspect.getsource(mod.handle_run_error)
|
||||
assert src.index("is_pool_outage(") < src.index("is_transient_capacity_error(e, extra_text=p_stderr_tail)"), "a veto goes at the top of the decision"
|
||||
|
||||
|
||||
def test_the_sentence_names_openswarm_pro():
|
||||
from backend.apps.agents.manager.streaming.provider_error_speech import ProviderError, user_facing_sentence
|
||||
err = ProviderError(kind="overloaded", subscription_spent=False, status=503, lane=None, reset_seconds=None, raw=P_POOL_503)
|
||||
assert user_facing_sentence(err, "sonnet-5").startswith("OpenSwarm Pro")
|
||||
Reference in New Issue
Block a user