From 86e3df51c3fca6e29b19f236557f6e5eed92034a Mon Sep 17 00:00:00 2001 From: ciregenz Date: Fri, 14 Aug 2026 17:35:27 -0700 Subject: [PATCH] [eric] agents: a core MCP server that never connects rebuilds the session instead of leaving it toolless (ENG-303) --- backend/apps/agents/manager/run/TurnRunner.py | 5 + .../manager/streaming/core_mcp_health.py | 82 ++++++++++++++++ backend/tests/test_core_mcp_health.py | 93 +++++++++++++++++++ 3 files changed, 180 insertions(+) create mode 100644 backend/apps/agents/manager/streaming/core_mcp_health.py create mode 100644 backend/tests/test_core_mcp_health.py diff --git a/backend/apps/agents/manager/run/TurnRunner.py b/backend/apps/agents/manager/run/TurnRunner.py index f8d770bd..8f63d212 100644 --- a/backend/apps/agents/manager/run/TurnRunner.py +++ b/backend/apps/agents/manager/run/TurnRunner.py @@ -18,6 +18,7 @@ from backend.apps.agents.manager.streaming.handle_stream_event import handle_str from backend.apps.agents.manager.streaming.handle_assistant_message import handle_assistant_message from backend.apps.agents.manager.streaming.handle_result_message import TurnResultError, handle_result_message from backend.apps.agents.manager.streaming.note_provider_retry import note_provider_retry, settle_provider_retries +from backend.apps.agents.manager.streaming.core_mcp_health import note_core_mcp_health from backend.apps.agents.manager.run.client_pool import ( SdkClientLike, acquire_client, @@ -119,6 +120,10 @@ class TurnRunner(AgentManagerProtocol): raw = message.__dict__ if hasattr(message, '__dict__') else str(message) logger.info(f"[MCP-DEBUG] SystemMessage: {raw}") p_subtype = getattr(message, "subtype", "") + if p_subtype == "init": + # The CLI states its MCP connection status here; a core server that never + # connected means a session with none of its tools, forever (ENG-303). + note_core_mcp_health(session, session_id, raw) if p_subtype == "compact_boundary": turn.compact_boundaries += 1 elif p_subtype == "api_retry": diff --git a/backend/apps/agents/manager/streaming/core_mcp_health.py b/backend/apps/agents/manager/streaming/core_mcp_health.py new file mode 100644 index 00000000..22f7b5ff --- /dev/null +++ b/backend/apps/agents/manager/streaming/core_mcp_health.py @@ -0,0 +1,82 @@ +"""Did the session's core MCP server actually connect? + +The watchdog next door handles a sidecar that freezes MID-CALL. This is the other half, and it is +the one that matches the original report: when the sidecar is unhealthy at CONNECT time, the tools +never register at all, so every `mcp__openswarm-core__*` call comes back "No such tool available" +INSTANTLY. Nothing hangs, no tool result is produced, no hook fires, and the agent spends the whole +session hunting for tools that were never there (measured 2026-08-14: 25 messages of Glob and Bash +before it gave up, and the fact it was asked to save was silently lost). + +The CLI tells us, every turn, in its init message: + + mcp_servers: [{'name': 'openswarm-core', 'status': 'connected'}] + +so this is a fact we can read rather than a symptom we have to infer. +""" + +import logging +from typing import Optional + +from typeguard import typechecked + +logger = logging.getLogger(__name__) + +CORE_SERVER = "openswarm-core" + + +@typechecked +def core_mcp_status(init_payload: object) -> Optional[str]: + """The reported status of our core server in an init message, or None when the message says + nothing about it (a different subtype, an older CLI, a shape we don't recognise). None means + "no opinion", never "broken": inventing a failure here would respawn healthy sessions.""" + data = init_payload + if isinstance(data, dict) and isinstance(data.get("data"), dict): + data = data["data"] + if not isinstance(data, dict): + return None + servers = data.get("mcp_servers") + if not isinstance(servers, list): + return None + for entry in servers: + if isinstance(entry, dict) and entry.get("name") == CORE_SERVER: + status = entry.get("status") + return str(status) if status is not None else None + return None + + +@typechecked +def core_mcp_failed_to_connect(init_payload: object) -> bool: + """True only when the CLI explicitly reports our core server as NOT connected. Silence, an + unknown shape, or a missing entry all read as False, so this can only ever fire on a positive + statement of failure.""" + status = core_mcp_status(init_payload) + return status is not None and status != "connected" + + +@typechecked +def note_core_mcp_health(session: object, session_id: str, init_payload: object) -> bool: + """Arm a fresh-session rebuild when the core server did not connect. The next turn respawns the + CLI (the machinery ENG-258 already uses for unclassified failures), which is the only cure: + MCP registration happens once at connect, so a toolless session stays toolless forever. + Returns whether it armed.""" + if not core_mcp_failed_to_connect(init_payload): + return False + status = core_mcp_status(init_payload) + logger.warning( + f"Agent {session_id}: core MCP server reported '{status}', not connected; " + "arming a fresh CLI session so the agent is not left without its tools" + ) + try: + session.needs_fresh_session = True # type: ignore[attr-defined] + except Exception: + return False + try: + from backend.apps.service.client import submit_diagnostic + submit_diagnostic({ + "kind": "core_mcp_not_connected", + "session_id": session_id, + "status": status, + }) + except Exception: + pass + return True diff --git a/backend/tests/test_core_mcp_health.py b/backend/tests/test_core_mcp_health.py new file mode 100644 index 00000000..bef262da --- /dev/null +++ b/backend/tests/test_core_mcp_health.py @@ -0,0 +1,93 @@ +"""The connect-time half of ENG-303 (found 2026-08-14 by freezing the sidecar BEFORE the CLI +connected). The mid-call watchdog cannot see this one: nothing hangs, so no tool result is +produced and no hook fires. Every `mcp__openswarm-core__*` call returns "No such tool available" +instantly and the session is toolless for its whole life, because MCP registration happens once. + +The CLI states the fact in its init message, so we read it instead of inferring it: + + mcp_servers: [{'name': 'openswarm-core', 'status': 'connected'}] + +The risk to guard is the opposite direction: a false positive respawns a HEALTHY session, so +anything other than an explicit non-connected status must read as "no opinion". +""" + +import inspect + +from backend.apps.agents.manager.streaming.core_mcp_health import ( + core_mcp_failed_to_connect, + core_mcp_status, + note_core_mcp_health, +) + + +def p_init(servers): + # The real shape, as logged live: subtype + a nested data dict. + return {"subtype": "init", "data": {"type": "system", "subtype": "init", "mcp_servers": servers}} + + +# --------------------------------------------------------------------- reading the fact + + +def test_a_connected_core_server_is_not_a_failure(): + payload = p_init([{"name": "openswarm-core", "status": "connected"}]) + assert core_mcp_status(payload) == "connected" + assert core_mcp_failed_to_connect(payload) is False + + +def test_an_explicitly_failed_core_server_is_caught(): + for bad in ("failed", "error", "disconnected", "pending"): + assert core_mcp_failed_to_connect(p_init([{"name": "openswarm-core", "status": bad}])) is True, bad + + +def test_the_flat_shape_works_too(): + # Some payloads arrive without the nested data wrapper. + assert core_mcp_failed_to_connect({"mcp_servers": [{"name": "openswarm-core", "status": "failed"}]}) is True + + +# --------------------------------------------------------------------- never invent a failure + + +def test_silence_is_not_failure(): + # No opinion must never respawn a healthy session. + assert core_mcp_failed_to_connect(p_init([])) is False + assert core_mcp_failed_to_connect({"subtype": "init", "data": {}}) is False + assert core_mcp_failed_to_connect({}) is False + assert core_mcp_failed_to_connect(None) is False + assert core_mcp_failed_to_connect("not a dict at all") is False + + +def test_another_server_failing_says_nothing_about_ours(): + payload = p_init([{"name": "some-other-mcp", "status": "failed"}]) + assert core_mcp_status(payload) is None + assert core_mcp_failed_to_connect(payload) is False + + +def test_a_malformed_entry_is_ignored_rather_than_read_as_broken(): + assert core_mcp_failed_to_connect(p_init(["not-a-dict", 7, None])) is False + assert core_mcp_failed_to_connect(p_init([{"name": "openswarm-core"}])) is False + + +# --------------------------------------------------------------------- the consequence + + +class P_Session: + needs_fresh_session = False + + +def test_a_failed_connect_arms_a_fresh_cli_session(): + s = P_Session() + assert note_core_mcp_health(s, "sess-1", p_init([{"name": "openswarm-core", "status": "failed"}])) is True + assert s.needs_fresh_session is True, "a toolless session stays toolless until the CLI respawns" + + +def test_a_healthy_connect_changes_nothing(): + s = P_Session() + assert note_core_mcp_health(s, "sess-2", p_init([{"name": "openswarm-core", "status": "connected"}])) is False + assert s.needs_fresh_session is False + + +def test_the_turn_runner_consults_this_on_init(): + from backend.apps.agents.manager.run import TurnRunner + src = inspect.getsource(TurnRunner) + assert "note_core_mcp_health" in src + assert 'p_subtype == "init"' in src, "the status is only reported on the init message"