[eric] agents: a core MCP server that never connects rebuilds the session instead of leaving it toolless (ENG-303)

This commit is contained in:
ciregenz
2026-08-14 17:35:27 -07:00
parent c11df8eea3
commit 86e3df51c3
3 changed files with 180 additions and 0 deletions
@@ -18,6 +18,7 @@ from backend.apps.agents.manager.streaming.handle_stream_event import handle_str
from backend.apps.agents.manager.streaming.handle_assistant_message import handle_assistant_message
from backend.apps.agents.manager.streaming.handle_result_message import TurnResultError, handle_result_message
from backend.apps.agents.manager.streaming.note_provider_retry import note_provider_retry, settle_provider_retries
from backend.apps.agents.manager.streaming.core_mcp_health import note_core_mcp_health
from backend.apps.agents.manager.run.client_pool import (
SdkClientLike,
acquire_client,
@@ -119,6 +120,10 @@ class TurnRunner(AgentManagerProtocol):
raw = message.__dict__ if hasattr(message, '__dict__') else str(message)
logger.info(f"[MCP-DEBUG] SystemMessage: {raw}")
p_subtype = getattr(message, "subtype", "")
if p_subtype == "init":
# The CLI states its MCP connection status here; a core server that never
# connected means a session with none of its tools, forever (ENG-303).
note_core_mcp_health(session, session_id, raw)
if p_subtype == "compact_boundary":
turn.compact_boundaries += 1
elif p_subtype == "api_retry":
@@ -0,0 +1,82 @@
"""Did the session's core MCP server actually connect?
The watchdog next door handles a sidecar that freezes MID-CALL. This is the other half, and it is
the one that matches the original report: when the sidecar is unhealthy at CONNECT time, the tools
never register at all, so every `mcp__openswarm-core__*` call comes back "No such tool available"
INSTANTLY. Nothing hangs, no tool result is produced, no hook fires, and the agent spends the whole
session hunting for tools that were never there (measured 2026-08-14: 25 messages of Glob and Bash
before it gave up, and the fact it was asked to save was silently lost).
The CLI tells us, every turn, in its init message:
mcp_servers: [{'name': 'openswarm-core', 'status': 'connected'}]
so this is a fact we can read rather than a symptom we have to infer.
"""
import logging
from typing import Optional
from typeguard import typechecked
logger = logging.getLogger(__name__)
CORE_SERVER = "openswarm-core"
@typechecked
def core_mcp_status(init_payload: object) -> Optional[str]:
"""The reported status of our core server in an init message, or None when the message says
nothing about it (a different subtype, an older CLI, a shape we don't recognise). None means
"no opinion", never "broken": inventing a failure here would respawn healthy sessions."""
data = init_payload
if isinstance(data, dict) and isinstance(data.get("data"), dict):
data = data["data"]
if not isinstance(data, dict):
return None
servers = data.get("mcp_servers")
if not isinstance(servers, list):
return None
for entry in servers:
if isinstance(entry, dict) and entry.get("name") == CORE_SERVER:
status = entry.get("status")
return str(status) if status is not None else None
return None
@typechecked
def core_mcp_failed_to_connect(init_payload: object) -> bool:
"""True only when the CLI explicitly reports our core server as NOT connected. Silence, an
unknown shape, or a missing entry all read as False, so this can only ever fire on a positive
statement of failure."""
status = core_mcp_status(init_payload)
return status is not None and status != "connected"
@typechecked
def note_core_mcp_health(session: object, session_id: str, init_payload: object) -> bool:
"""Arm a fresh-session rebuild when the core server did not connect. The next turn respawns the
CLI (the machinery ENG-258 already uses for unclassified failures), which is the only cure:
MCP registration happens once at connect, so a toolless session stays toolless forever.
Returns whether it armed."""
if not core_mcp_failed_to_connect(init_payload):
return False
status = core_mcp_status(init_payload)
logger.warning(
f"Agent {session_id}: core MCP server reported '{status}', not connected; "
"arming a fresh CLI session so the agent is not left without its tools"
)
try:
session.needs_fresh_session = True # type: ignore[attr-defined]
except Exception:
return False
try:
from backend.apps.service.client import submit_diagnostic
submit_diagnostic({
"kind": "core_mcp_not_connected",
"session_id": session_id,
"status": status,
})
except Exception:
pass
return True
+93
View File
@@ -0,0 +1,93 @@
"""The connect-time half of ENG-303 (found 2026-08-14 by freezing the sidecar BEFORE the CLI
connected). The mid-call watchdog cannot see this one: nothing hangs, so no tool result is
produced and no hook fires. Every `mcp__openswarm-core__*` call returns "No such tool available"
instantly and the session is toolless for its whole life, because MCP registration happens once.
The CLI states the fact in its init message, so we read it instead of inferring it:
mcp_servers: [{'name': 'openswarm-core', 'status': 'connected'}]
The risk to guard is the opposite direction: a false positive respawns a HEALTHY session, so
anything other than an explicit non-connected status must read as "no opinion".
"""
import inspect
from backend.apps.agents.manager.streaming.core_mcp_health import (
core_mcp_failed_to_connect,
core_mcp_status,
note_core_mcp_health,
)
def p_init(servers):
# The real shape, as logged live: subtype + a nested data dict.
return {"subtype": "init", "data": {"type": "system", "subtype": "init", "mcp_servers": servers}}
# --------------------------------------------------------------------- reading the fact
def test_a_connected_core_server_is_not_a_failure():
payload = p_init([{"name": "openswarm-core", "status": "connected"}])
assert core_mcp_status(payload) == "connected"
assert core_mcp_failed_to_connect(payload) is False
def test_an_explicitly_failed_core_server_is_caught():
for bad in ("failed", "error", "disconnected", "pending"):
assert core_mcp_failed_to_connect(p_init([{"name": "openswarm-core", "status": bad}])) is True, bad
def test_the_flat_shape_works_too():
# Some payloads arrive without the nested data wrapper.
assert core_mcp_failed_to_connect({"mcp_servers": [{"name": "openswarm-core", "status": "failed"}]}) is True
# --------------------------------------------------------------------- never invent a failure
def test_silence_is_not_failure():
# No opinion must never respawn a healthy session.
assert core_mcp_failed_to_connect(p_init([])) is False
assert core_mcp_failed_to_connect({"subtype": "init", "data": {}}) is False
assert core_mcp_failed_to_connect({}) is False
assert core_mcp_failed_to_connect(None) is False
assert core_mcp_failed_to_connect("not a dict at all") is False
def test_another_server_failing_says_nothing_about_ours():
payload = p_init([{"name": "some-other-mcp", "status": "failed"}])
assert core_mcp_status(payload) is None
assert core_mcp_failed_to_connect(payload) is False
def test_a_malformed_entry_is_ignored_rather_than_read_as_broken():
assert core_mcp_failed_to_connect(p_init(["not-a-dict", 7, None])) is False
assert core_mcp_failed_to_connect(p_init([{"name": "openswarm-core"}])) is False
# --------------------------------------------------------------------- the consequence
class P_Session:
needs_fresh_session = False
def test_a_failed_connect_arms_a_fresh_cli_session():
s = P_Session()
assert note_core_mcp_health(s, "sess-1", p_init([{"name": "openswarm-core", "status": "failed"}])) is True
assert s.needs_fresh_session is True, "a toolless session stays toolless until the CLI respawns"
def test_a_healthy_connect_changes_nothing():
s = P_Session()
assert note_core_mcp_health(s, "sess-2", p_init([{"name": "openswarm-core", "status": "connected"}])) is False
assert s.needs_fresh_session is False
def test_the_turn_runner_consults_this_on_init():
from backend.apps.agents.manager.run import TurnRunner
src = inspect.getsource(TurnRunner)
assert "note_core_mcp_health" in src
assert 'p_subtype == "init"' in src, "the status is only reported on the init message"