mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-28 12:34:50 +02:00
[eric] agents: a core MCP server that never connects rebuilds the session instead of leaving it toolless (ENG-303)
This commit is contained in:
@@ -18,6 +18,7 @@ from backend.apps.agents.manager.streaming.handle_stream_event import handle_str
|
||||
from backend.apps.agents.manager.streaming.handle_assistant_message import handle_assistant_message
|
||||
from backend.apps.agents.manager.streaming.handle_result_message import TurnResultError, handle_result_message
|
||||
from backend.apps.agents.manager.streaming.note_provider_retry import note_provider_retry, settle_provider_retries
|
||||
from backend.apps.agents.manager.streaming.core_mcp_health import note_core_mcp_health
|
||||
from backend.apps.agents.manager.run.client_pool import (
|
||||
SdkClientLike,
|
||||
acquire_client,
|
||||
@@ -119,6 +120,10 @@ class TurnRunner(AgentManagerProtocol):
|
||||
raw = message.__dict__ if hasattr(message, '__dict__') else str(message)
|
||||
logger.info(f"[MCP-DEBUG] SystemMessage: {raw}")
|
||||
p_subtype = getattr(message, "subtype", "")
|
||||
if p_subtype == "init":
|
||||
# The CLI states its MCP connection status here; a core server that never
|
||||
# connected means a session with none of its tools, forever (ENG-303).
|
||||
note_core_mcp_health(session, session_id, raw)
|
||||
if p_subtype == "compact_boundary":
|
||||
turn.compact_boundaries += 1
|
||||
elif p_subtype == "api_retry":
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Did the session's core MCP server actually connect?
|
||||
|
||||
The watchdog next door handles a sidecar that freezes MID-CALL. This is the other half, and it is
|
||||
the one that matches the original report: when the sidecar is unhealthy at CONNECT time, the tools
|
||||
never register at all, so every `mcp__openswarm-core__*` call comes back "No such tool available"
|
||||
INSTANTLY. Nothing hangs, no tool result is produced, no hook fires, and the agent spends the whole
|
||||
session hunting for tools that were never there (measured 2026-08-14: 25 messages of Glob and Bash
|
||||
before it gave up, and the fact it was asked to save was silently lost).
|
||||
|
||||
The CLI tells us, every turn, in its init message:
|
||||
|
||||
mcp_servers: [{'name': 'openswarm-core', 'status': 'connected'}]
|
||||
|
||||
so this is a fact we can read rather than a symptom we have to infer.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
from typeguard import typechecked
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
CORE_SERVER = "openswarm-core"
|
||||
|
||||
|
||||
@typechecked
|
||||
def core_mcp_status(init_payload: object) -> Optional[str]:
|
||||
"""The reported status of our core server in an init message, or None when the message says
|
||||
nothing about it (a different subtype, an older CLI, a shape we don't recognise). None means
|
||||
"no opinion", never "broken": inventing a failure here would respawn healthy sessions."""
|
||||
data = init_payload
|
||||
if isinstance(data, dict) and isinstance(data.get("data"), dict):
|
||||
data = data["data"]
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
servers = data.get("mcp_servers")
|
||||
if not isinstance(servers, list):
|
||||
return None
|
||||
for entry in servers:
|
||||
if isinstance(entry, dict) and entry.get("name") == CORE_SERVER:
|
||||
status = entry.get("status")
|
||||
return str(status) if status is not None else None
|
||||
return None
|
||||
|
||||
|
||||
@typechecked
|
||||
def core_mcp_failed_to_connect(init_payload: object) -> bool:
|
||||
"""True only when the CLI explicitly reports our core server as NOT connected. Silence, an
|
||||
unknown shape, or a missing entry all read as False, so this can only ever fire on a positive
|
||||
statement of failure."""
|
||||
status = core_mcp_status(init_payload)
|
||||
return status is not None and status != "connected"
|
||||
|
||||
|
||||
@typechecked
|
||||
def note_core_mcp_health(session: object, session_id: str, init_payload: object) -> bool:
|
||||
"""Arm a fresh-session rebuild when the core server did not connect. The next turn respawns the
|
||||
CLI (the machinery ENG-258 already uses for unclassified failures), which is the only cure:
|
||||
MCP registration happens once at connect, so a toolless session stays toolless forever.
|
||||
Returns whether it armed."""
|
||||
if not core_mcp_failed_to_connect(init_payload):
|
||||
return False
|
||||
status = core_mcp_status(init_payload)
|
||||
logger.warning(
|
||||
f"Agent {session_id}: core MCP server reported '{status}', not connected; "
|
||||
"arming a fresh CLI session so the agent is not left without its tools"
|
||||
)
|
||||
try:
|
||||
session.needs_fresh_session = True # type: ignore[attr-defined]
|
||||
except Exception:
|
||||
return False
|
||||
try:
|
||||
from backend.apps.service.client import submit_diagnostic
|
||||
submit_diagnostic({
|
||||
"kind": "core_mcp_not_connected",
|
||||
"session_id": session_id,
|
||||
"status": status,
|
||||
})
|
||||
except Exception:
|
||||
pass
|
||||
return True
|
||||
@@ -0,0 +1,93 @@
|
||||
"""The connect-time half of ENG-303 (found 2026-08-14 by freezing the sidecar BEFORE the CLI
|
||||
connected). The mid-call watchdog cannot see this one: nothing hangs, so no tool result is
|
||||
produced and no hook fires. Every `mcp__openswarm-core__*` call returns "No such tool available"
|
||||
instantly and the session is toolless for its whole life, because MCP registration happens once.
|
||||
|
||||
The CLI states the fact in its init message, so we read it instead of inferring it:
|
||||
|
||||
mcp_servers: [{'name': 'openswarm-core', 'status': 'connected'}]
|
||||
|
||||
The risk to guard is the opposite direction: a false positive respawns a HEALTHY session, so
|
||||
anything other than an explicit non-connected status must read as "no opinion".
|
||||
"""
|
||||
|
||||
import inspect
|
||||
|
||||
from backend.apps.agents.manager.streaming.core_mcp_health import (
|
||||
core_mcp_failed_to_connect,
|
||||
core_mcp_status,
|
||||
note_core_mcp_health,
|
||||
)
|
||||
|
||||
|
||||
def p_init(servers):
|
||||
# The real shape, as logged live: subtype + a nested data dict.
|
||||
return {"subtype": "init", "data": {"type": "system", "subtype": "init", "mcp_servers": servers}}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------- reading the fact
|
||||
|
||||
|
||||
def test_a_connected_core_server_is_not_a_failure():
|
||||
payload = p_init([{"name": "openswarm-core", "status": "connected"}])
|
||||
assert core_mcp_status(payload) == "connected"
|
||||
assert core_mcp_failed_to_connect(payload) is False
|
||||
|
||||
|
||||
def test_an_explicitly_failed_core_server_is_caught():
|
||||
for bad in ("failed", "error", "disconnected", "pending"):
|
||||
assert core_mcp_failed_to_connect(p_init([{"name": "openswarm-core", "status": bad}])) is True, bad
|
||||
|
||||
|
||||
def test_the_flat_shape_works_too():
|
||||
# Some payloads arrive without the nested data wrapper.
|
||||
assert core_mcp_failed_to_connect({"mcp_servers": [{"name": "openswarm-core", "status": "failed"}]}) is True
|
||||
|
||||
|
||||
# --------------------------------------------------------------------- never invent a failure
|
||||
|
||||
|
||||
def test_silence_is_not_failure():
|
||||
# No opinion must never respawn a healthy session.
|
||||
assert core_mcp_failed_to_connect(p_init([])) is False
|
||||
assert core_mcp_failed_to_connect({"subtype": "init", "data": {}}) is False
|
||||
assert core_mcp_failed_to_connect({}) is False
|
||||
assert core_mcp_failed_to_connect(None) is False
|
||||
assert core_mcp_failed_to_connect("not a dict at all") is False
|
||||
|
||||
|
||||
def test_another_server_failing_says_nothing_about_ours():
|
||||
payload = p_init([{"name": "some-other-mcp", "status": "failed"}])
|
||||
assert core_mcp_status(payload) is None
|
||||
assert core_mcp_failed_to_connect(payload) is False
|
||||
|
||||
|
||||
def test_a_malformed_entry_is_ignored_rather_than_read_as_broken():
|
||||
assert core_mcp_failed_to_connect(p_init(["not-a-dict", 7, None])) is False
|
||||
assert core_mcp_failed_to_connect(p_init([{"name": "openswarm-core"}])) is False
|
||||
|
||||
|
||||
# --------------------------------------------------------------------- the consequence
|
||||
|
||||
|
||||
class P_Session:
|
||||
needs_fresh_session = False
|
||||
|
||||
|
||||
def test_a_failed_connect_arms_a_fresh_cli_session():
|
||||
s = P_Session()
|
||||
assert note_core_mcp_health(s, "sess-1", p_init([{"name": "openswarm-core", "status": "failed"}])) is True
|
||||
assert s.needs_fresh_session is True, "a toolless session stays toolless until the CLI respawns"
|
||||
|
||||
|
||||
def test_a_healthy_connect_changes_nothing():
|
||||
s = P_Session()
|
||||
assert note_core_mcp_health(s, "sess-2", p_init([{"name": "openswarm-core", "status": "connected"}])) is False
|
||||
assert s.needs_fresh_session is False
|
||||
|
||||
|
||||
def test_the_turn_runner_consults_this_on_init():
|
||||
from backend.apps.agents.manager.run import TurnRunner
|
||||
src = inspect.getsource(TurnRunner)
|
||||
assert "note_core_mcp_health" in src
|
||||
assert 'p_subtype == "init"' in src, "the status is only reported on the init message"
|
||||
Reference in New Issue
Block a user