From b90e5c6353483b561b2fe6c62333a51a4053fd08 Mon Sep 17 00:00:00 2001 From: haikdc Date: Mon, 16 Mar 2026 05:01:56 -0700 Subject: [PATCH] [Haik]: (Practically all forms of dashboard interaction glitches) (made dashboard selection process much more intuitive and seamless - and high level) (added BrowserAgent so the outer agent can delegate more complex tasks) (added tool to parallelize across multiple browser agents since the abstraction keeps the tool calls so high level) (Added a temporary agent insight ui over each browser as it does stuff - human in the loop is still needed) --- backend/apps/agents/agent_manager.py | 88 ++-- backend/apps/agents/browser_agent.py | 464 ++++++++++++++++++ .../apps/agents/browser_agent_mcp_server.py | 288 +++++++++++ backend/apps/agents/models.py | 1 + backend/main.py | 30 ++ .../app/components/useDomElementSelector.ts | 12 +- .../src/app/pages/AgentChat/ChatInput.tsx | 34 +- .../src/app/pages/Dashboard/AgentCard.tsx | 23 + .../pages/Dashboard/BrowserAgentOverlay.tsx | 327 ++++++++++++ .../src/app/pages/Dashboard/BrowserCard.tsx | 48 +- .../src/app/pages/Dashboard/Dashboard.tsx | 17 +- .../app/pages/Dashboard/DashboardViewCard.tsx | 22 + .../pages/Dashboard/useDashboardSelection.ts | 2 + frontend/src/app/pages/Settings/Settings.tsx | 36 +- frontend/src/shared/state/agentsSlice.ts | 1 + .../src/shared/state/dashboardLayoutSlice.ts | 11 + frontend/src/shared/state/settingsSlice.ts | 3 +- frontend/src/shared/ws/WebSocketManager.ts | 7 + 18 files changed, 1326 insertions(+), 88 deletions(-) create mode 100644 backend/apps/agents/browser_agent.py create mode 100644 backend/apps/agents/browser_agent_mcp_server.py create mode 100644 frontend/src/app/pages/Dashboard/BrowserAgentOverlay.tsx diff --git a/backend/apps/agents/agent_manager.py b/backend/apps/agents/agent_manager.py index 563de372..c92b6b10 100644 --- a/backend/apps/agents/agent_manager.py +++ b/backend/apps/agents/agent_manager.py @@ -233,7 +233,7 @@ class AgentManager: ) def _build_browser_context(self, dashboard_id: str | None) -> str | None: - """Build a context block listing browser cards on the agent's dashboard.""" + """Build a context block listing browser cards and delegation instructions.""" if not dashboard_id: return None try: @@ -243,36 +243,54 @@ class AgentManager: return None raw = dashboard.model_dump(mode="json") browser_cards = raw.get("layout", {}).get("browser_cards", {}) - if not browser_cards: - return None lines = [ - "", - "The following browser cards are on the dashboard. " - "Use the browser_id value when calling any Browser tool (BrowserNavigate, " - "BrowserClick, BrowserType, BrowserScreenshot, BrowserGetText, BrowserGetElements, " - "BrowserEvaluate).\n", + "", + "You have access to browser automation through the BrowserAgent and BrowserAgents tools.", + "", + "- **BrowserAgent(task, browser_id?, url?)**: Delegate a single browser task to a dedicated browser agent. " + "The browser agent will autonomously navigate, click, type, and interact with the page, then return a summary and screenshot.", + "- **BrowserAgents(tasks)**: Run multiple browser tasks in parallel, each on a different browser.", + "", + "You do NOT have direct access to low-level browser tools (click, type, screenshot, etc.). " + "Instead, describe what you want accomplished and the browser agent will handle the details.", + "", + "If you omit browser_id, a new browser card will be auto-created. " + "If you provide a url without a browser_id, the new browser navigates there first.", ] - for card in browser_cards.values(): - bid = card.get("browser_id", "") - tabs = card.get("tabs", []) - active_tab_id = card.get("activeTabId", "") - active_tab = next((t for t in tabs if t.get("id") == active_tab_id), None) - url = (active_tab or {}).get("url", card.get("url", "")) - title = (active_tab or {}).get("title", "") - lines.append(f"- browser_id: \"{bid}\"") - if title: - lines.append(f" Title: {title}") - if url: - lines.append(f" URL: {url}") - if len(tabs) > 1: - lines.append(f" Tabs ({len(tabs)}):") - for t in tabs: - marker = " (active)" if t.get("id") == active_tab_id else "" - lines.append(f" - tab_id: \"{t.get('id', '')}\" | {t.get('title') or t.get('url', '')}{marker}") - lines.append("") + + if browser_cards: + lines.append("") + lines.append("Available browser cards on the dashboard:") + for card in browser_cards.values(): + bid = card.get("browser_id", "") + tabs = card.get("tabs", []) + active_tab_id = card.get("activeTabId", "") + active_tab = next((t for t in tabs if t.get("id") == active_tab_id), None) + url = (active_tab or {}).get("url", card.get("url", "")) + title = (active_tab or {}).get("title", "") + lines.append(f"- browser_id: \"{bid}\"") + if title: + lines.append(f" Title: {title}") + if url: + lines.append(f" URL: {url}") + + lines.append("") return "\n".join(lines) + def _get_pre_selected_browser_ids(self, dashboard_id: str | None) -> list[str]: + """Return browser_ids of all browser cards currently on the dashboard.""" + if not dashboard_id: + return [] + try: + from backend.apps.dashboards.dashboards import _load as load_dashboard + dashboard = load_dashboard(dashboard_id) + except Exception: + return [] + raw = dashboard.model_dump(mode="json") + browser_cards = raw.get("layout", {}).get("browser_cards", {}) + return [card.get("browser_id", "") for card in browser_cards.values() if card.get("browser_id")] + def _compose_system_prompt(self, default_prompt: str | None, mode_prompt: str | None, session_prompt: str | None, connected_tools_ctx: str | None = None, outputs_ctx: str | None = None, browser_ctx: str | None = None) -> str | None: parts = [p for p in (default_prompt, mode_prompt, session_prompt, connected_tools_ctx, outputs_ctx, browser_ctx) if p] return "\n\n".join(parts) if parts else None @@ -590,14 +608,20 @@ class AgentManager: mcp_servers = await self._build_mcp_servers(session.allowed_tools) - browser_server_path = os.path.join( - os.path.dirname(__file__), "browser_mcp_server.py" + browser_agent_server_path = os.path.join( + os.path.dirname(__file__), "browser_agent_mcp_server.py" ) backend_port = os.environ.get("OPENSWARM_PORT", "8324") - mcp_servers["openswarm-browser"] = { + pre_selected_bids = self._get_pre_selected_browser_ids(session.dashboard_id) + mcp_servers["openswarm-browser-agent"] = { "command": sys.executable, - "args": [browser_server_path], - "env": {"OPENSWARM_PORT": backend_port}, + "args": [browser_agent_server_path], + "env": { + "OPENSWARM_PORT": backend_port, + "OPENSWARM_AGENT_MODEL": session.model, + "OPENSWARM_DASHBOARD_ID": session.dashboard_id or "", + "OPENSWARM_PRE_SELECTED_BROWSER_IDS": ",".join(pre_selected_bids), + }, "type": "stdio", } @@ -623,7 +647,7 @@ class AgentManager: else: effective_allowed.append(f"mcp__{name}__*") - effective_allowed.append("mcp__openswarm-browser__*") + effective_allowed.append("mcp__openswarm-browser-agent__*") options_kwargs = { "model": session.model, diff --git a/backend/apps/agents/browser_agent.py b/backend/apps/agents/browser_agent.py new file mode 100644 index 00000000..11244acf --- /dev/null +++ b/backend/apps/agents/browser_agent.py @@ -0,0 +1,464 @@ +""" +Browser sub-agent runner. + +Provides a lightweight Anthropic API tool-use loop that drives browser +interactions directly through ws_manager (no MCP subprocess needed). +Sub-agents appear as visible AgentSession cards on the dashboard. +""" + +import asyncio +import json +import logging +import time +from datetime import datetime +from uuid import uuid4 + +import anthropic + +from backend.apps.agents.models import AgentSession, Message +from backend.apps.agents.ws_manager import ws_manager + +logger = logging.getLogger(__name__) + +MODEL_MAP = { + "sonnet": "claude-sonnet-4-20250514", + "opus": "claude-opus-4-20250514", + "haiku": "claude-haiku-4-20250414", +} + +BROWSER_TOOLS_SCHEMA = [ + { + "name": "BrowserScreenshot", + "description": ( + "Capture a screenshot of the browser page. Returns the screenshot as a " + "base64-encoded PNG image. Use this to see what is currently displayed." + ), + "input_schema": { + "type": "object", + "properties": {}, + "required": [], + }, + }, + { + "name": "BrowserGetText", + "description": ( + "Get the visible text content of the browser page. Returns up to 15000 characters." + ), + "input_schema": { + "type": "object", + "properties": {}, + "required": [], + }, + }, + { + "name": "BrowserNavigate", + "description": "Navigate the browser to a URL.", + "input_schema": { + "type": "object", + "properties": { + "url": {"type": "string", "description": "The URL to navigate to."}, + }, + "required": ["url"], + }, + }, + { + "name": "BrowserClick", + "description": "Click an element identified by a CSS selector. Use BrowserGetElements first to discover valid selectors.", + "input_schema": { + "type": "object", + "properties": { + "selector": {"type": "string", "description": "CSS selector of the element to click."}, + }, + "required": ["selector"], + }, + }, + { + "name": "BrowserType", + "description": "Type text into an input element. Clears existing value first.", + "input_schema": { + "type": "object", + "properties": { + "selector": {"type": "string", "description": "CSS selector of the input element."}, + "text": {"type": "string", "description": "The text to type."}, + }, + "required": ["selector", "text"], + }, + }, + { + "name": "BrowserEvaluate", + "description": "Evaluate a JavaScript expression in the browser page and return the result.", + "input_schema": { + "type": "object", + "properties": { + "expression": {"type": "string", "description": "JavaScript expression to evaluate."}, + }, + "required": ["expression"], + }, + }, + { + "name": "BrowserGetElements", + "description": ( + "Get a list of interactive elements on the page with CSS selectors. " + "Call this BEFORE clicking or typing so you know which selectors are valid." + ), + "input_schema": { + "type": "object", + "properties": { + "selector": { + "type": "string", + "description": "Optional CSS selector to scope the search (e.g. 'form', '#main'). Defaults to 'body'.", + }, + }, + "required": [], + }, + }, +] + +ACTION_MAP = { + "BrowserScreenshot": "screenshot", + "BrowserGetText": "get_text", + "BrowserNavigate": "navigate", + "BrowserClick": "click", + "BrowserType": "type", + "BrowserEvaluate": "evaluate", + "BrowserGetElements": "get_elements", +} + +SYSTEM_PROMPT = ( + "You are a browser automation agent. You control a single browser tab and " + "execute the task you are given.\n\n" + "Strategy:\n" + "1. Start by taking a screenshot or calling BrowserGetElements to understand the page.\n" + "2. Use BrowserGetElements BEFORE clicking or typing to discover valid CSS selectors.\n" + "3. After performing actions, take a screenshot to verify the result.\n" + "4. If an action fails, try alternative selectors or approaches.\n" + "5. When the task is complete, provide a clear summary of what you accomplished.\n\n" + "You have access ONLY to browser tools. Do not ask the user questions — " + "complete the task autonomously to the best of your ability." +) + +MAX_TURNS = 25 + + +async def execute_browser_tool( + tool_name: str, tool_input: dict, browser_id: str, tab_id: str = "", +) -> dict: + """Execute a browser tool via ws_manager directly (no MCP/HTTP round-trip).""" + action = ACTION_MAP.get(tool_name) + if not action: + return {"error": f"Unknown browser tool: {tool_name}"} + + params = {k: v for k, v in tool_input.items()} + request_id = uuid4().hex + result = await ws_manager.send_browser_command( + request_id, action, browser_id, params, tab_id=tab_id, + ) + return result + + +def _format_tool_result(result: dict, tool_name: str) -> list[dict]: + """Convert a browser command result dict into Anthropic API content blocks.""" + if "error" in result: + return [{"type": "text", "text": f"Error: {result['error']}"}] + + if tool_name == "BrowserScreenshot" and result.get("image"): + blocks = [ + { + "type": "image", + "source": { + "type": "base64", + "media_type": "image/png", + "data": result["image"], + }, + }, + {"type": "text", "text": f"Screenshot captured. URL: {result.get('url', 'unknown')}"}, + ] + return blocks + + text = result.get("text", json.dumps(result)) + return [{"type": "text", "text": str(text)}] + + +async def run_browser_agent( + task: str, + browser_id: str, + model: str, + api_key: str, + dashboard_id: str | None = None, + tab_id: str = "", + pre_selected: bool = False, + initial_url: str | None = None, +) -> dict: + """Run a browser sub-agent loop for a single browser card. + + Creates a visible AgentSession, streams progress via WebSocket, + and returns the full action log + summary + final screenshot. + """ + from backend.apps.agents.agent_manager import agent_manager + + session_id = uuid4().hex + session = AgentSession( + id=session_id, + name=f"Browser Agent", + model=model, + mode="browser-agent", + status="running", + dashboard_id=dashboard_id, + browser_id=browser_id, + system_prompt=SYSTEM_PROMPT, + ) + agent_manager.sessions[session_id] = session + + await ws_manager.send_to_session(session_id, "agent:status", { + "session_id": session_id, + "status": "running", + "session": session.model_dump(mode="json"), + }) + + if initial_url: + nav_result = await execute_browser_tool( + "BrowserNavigate", {"url": initial_url}, browser_id, tab_id, + ) + logger.info(f"Browser agent {session_id}: navigated to {initial_url}: {nav_result.get('text', nav_result.get('error', ''))}") + + api_model = MODEL_MAP.get(model, model) + client = anthropic.AsyncAnthropic(api_key=api_key) + + messages: list[dict] = [{"role": "user", "content": task}] + action_log: list[dict] = [] + final_screenshot: str | None = None + + user_msg = Message(role="user", content=task) + session.messages.append(user_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": user_msg.model_dump(mode="json"), + }) + + try: + for turn in range(MAX_TURNS): + response = await client.messages.create( + model=api_model, + max_tokens=4096, + system=SYSTEM_PROMPT, + tools=BROWSER_TOOLS_SCHEMA, + messages=messages, + ) + + assistant_content = [] + text_parts = [] + tool_uses = [] + + for block in response.content: + if block.type == "text": + text_parts.append(block.text) + assistant_content.append({"type": "text", "text": block.text}) + elif block.type == "tool_use": + tool_uses.append(block) + assistant_content.append({ + "type": "tool_use", + "id": block.id, + "name": block.name, + "input": block.input, + }) + + if text_parts: + asst_msg = Message( + role="assistant", + content="\n".join(text_parts), + ) + session.messages.append(asst_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": asst_msg.model_dump(mode="json"), + }) + + for tu in tool_uses: + tool_msg = Message( + role="tool_call", + content={"id": tu.id, "tool": tu.name, "input": tu.input}, + ) + session.messages.append(tool_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": tool_msg.model_dump(mode="json"), + }) + + messages.append({"role": "assistant", "content": assistant_content}) + + if response.stop_reason != "tool_use": + break + + tool_results = [] + for tu in tool_uses: + start = time.time() + result = await execute_browser_tool( + tu.name, tu.input, browser_id, tab_id, + ) + elapsed_ms = int((time.time() - start) * 1000) + + action_log.append({ + "tool": tu.name, + "input": tu.input, + "result_summary": result.get("text", result.get("error", ""))[:200], + "elapsed_ms": elapsed_ms, + }) + + if tu.name == "BrowserScreenshot" and result.get("image"): + final_screenshot = result["image"] + + content_blocks = _format_tool_result(result, tu.name) + tool_results.append({ + "type": "tool_result", + "tool_use_id": tu.id, + "content": content_blocks, + }) + + result_text = result.get("text", result.get("error", "")) + result_msg = Message( + role="tool_result", + content={"text": result_text, "tool_name": tu.name, "elapsed_ms": elapsed_ms}, + ) + session.messages.append(result_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": result_msg.model_dump(mode="json"), + }) + + messages.append({"role": "user", "content": tool_results}) + + summary_parts = text_parts if text_parts else ["Task completed."] + summary = "\n".join(summary_parts) + + if not final_screenshot: + try: + ss_result = await execute_browser_tool( + "BrowserScreenshot", {}, browser_id, tab_id, + ) + if ss_result.get("image"): + final_screenshot = ss_result["image"] + except Exception: + pass + + session.status = "completed" + await ws_manager.send_to_session(session_id, "agent:status", { + "session_id": session_id, + "status": "completed", + "session": session.model_dump(mode="json"), + }) + + await asyncio.sleep(2.5) + try: + await agent_manager.close_session(session_id) + except Exception: + logger.warning(f"Failed to auto-close browser agent session {session_id}") + + return { + "session_id": session_id, + "browser_id": browser_id, + "summary": summary, + "action_log": action_log, + "final_screenshot": final_screenshot, + } + + except Exception as e: + logger.exception(f"Browser agent {session_id} error: {e}") + session.status = "error" + error_msg = Message(role="system", content=f"Error: {str(e)}") + session.messages.append(error_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": error_msg.model_dump(mode="json"), + }) + await ws_manager.send_to_session(session_id, "agent:status", { + "session_id": session_id, + "status": "error", + "session": session.model_dump(mode="json"), + }) + + await asyncio.sleep(2.5) + try: + await agent_manager.close_session(session_id) + except Exception: + logger.warning(f"Failed to auto-close browser agent session {session_id} after error") + + return { + "session_id": session_id, + "browser_id": browser_id, + "summary": f"Error: {str(e)}", + "action_log": action_log, + "final_screenshot": None, + } + + +async def _create_browser_card(dashboard_id: str, url: str) -> str: + """Create a new browser card on the dashboard and return its browser_id.""" + from backend.apps.dashboards.dashboards import _load, _save + from backend.apps.dashboards.models import BrowserCardPosition, BrowserTab + + dashboard = _load(dashboard_id) + browser_id = f"browser-{uuid4().hex[:8]}" + tab_id = f"tab-{uuid4().hex[:8]}" + tab = BrowserTab(id=tab_id, url=url or "https://www.google.com", title="") + card = BrowserCardPosition( + browser_id=browser_id, + url=url or "https://www.google.com", + tabs=[tab], + activeTabId=tab_id, + x=40, + y=100, + ) + dashboard.layout.browser_cards[browser_id] = card + dashboard.updated_at = datetime.now() + _save(dashboard) + + await ws_manager.broadcast_global("dashboard:browser_card_added", { + "dashboard_id": dashboard_id, + "browser_card": card.model_dump(mode="json"), + }) + return browser_id + + +async def run_browser_agents( + tasks: list[dict], + model: str, + api_key: str, + dashboard_id: str | None = None, + pre_selected_browser_ids: list[str] | None = None, +) -> list[dict]: + """Run multiple browser sub-agents in parallel. + + Each task dict has: { browser_id (optional), task, url (optional) } + Returns a list of result dicts, one per task. + """ + pre_selected = set(pre_selected_browser_ids or []) + + async def _run_one(task_def: dict) -> dict: + browser_id = task_def.get("browser_id", "") + task_text = task_def.get("task", "") + url = task_def.get("url", "") + + if not browser_id and dashboard_id: + browser_id = await _create_browser_card(dashboard_id, url) + await asyncio.sleep(2.0) + + is_pre_selected = browser_id in pre_selected + return await run_browser_agent( + task=task_text, + browser_id=browser_id, + model=model, + api_key=api_key, + dashboard_id=dashboard_id, + pre_selected=is_pre_selected, + initial_url=url if url and browser_id not in pre_selected else None, + ) + + results = await asyncio.gather(*[_run_one(t) for t in tasks], return_exceptions=True) + + final = [] + for r in results: + if isinstance(r, Exception): + final.append({"summary": f"Error: {str(r)}", "action_log": [], "final_screenshot": None}) + else: + final.append(r) + return final diff --git a/backend/apps/agents/browser_agent_mcp_server.py b/backend/apps/agents/browser_agent_mcp_server.py new file mode 100644 index 00000000..2a46a139 --- /dev/null +++ b/backend/apps/agents/browser_agent_mcp_server.py @@ -0,0 +1,288 @@ +#!/usr/bin/env python3 +""" +Stdio MCP server that exposes BrowserAgent and BrowserAgents delegation tools. + +Launched as a subprocess by the Claude Agent SDK. Proxies task delegation +to the OpenSwarm backend via HTTP, which runs browser sub-agents. +""" + +import base64 +import json +import sys +import os +import urllib.request +import urllib.error +from io import BytesIO + +try: + from PIL import Image + HAS_PIL = True +except ImportError: + HAS_PIL = False + +BACKEND_PORT = os.environ.get("OPENSWARM_PORT", "8324") +BACKEND_URL = f"http://127.0.0.1:{BACKEND_PORT}/api/browser-agent/run" +MODEL = os.environ.get("OPENSWARM_AGENT_MODEL", "sonnet") +DASHBOARD_ID = os.environ.get("OPENSWARM_DASHBOARD_ID", "") +PRE_SELECTED_BROWSER_IDS = os.environ.get("OPENSWARM_PRE_SELECTED_BROWSER_IDS", "") + +TOOLS = [ + { + "name": "BrowserAgent", + "description": ( + "Delegate a browser task to a dedicated browser agent. The browser agent " + "will autonomously perform the task (navigating, clicking, typing, etc.) " + "and return a summary of actions taken plus a final screenshot. " + "Use this for any task that requires interacting with a web page." + ), + "inputSchema": { + "type": "object", + "properties": { + "browser_id": { + "type": "string", + "description": ( + "The ID of the browser card to use. If omitted, a new browser " + "card will be automatically created." + ), + }, + "task": { + "type": "string", + "description": ( + "The task for the browser agent to perform. Be specific and " + "detailed about what you want accomplished." + ), + }, + "url": { + "type": "string", + "description": ( + "Optional starting URL. If provided and no browser_id is given, " + "the new browser will navigate here first." + ), + }, + }, + "required": ["task"], + }, + }, + { + "name": "BrowserAgents", + "description": ( + "Delegate multiple browser tasks to run in parallel, each on a different " + "browser. All tasks execute concurrently and results are returned together. " + "Use this when you need to perform tasks on multiple web pages simultaneously." + ), + "inputSchema": { + "type": "object", + "properties": { + "tasks": { + "type": "array", + "description": "Array of browser tasks to run in parallel.", + "items": { + "type": "object", + "properties": { + "browser_id": { + "type": "string", + "description": "Optional browser card ID. If omitted, a new browser will be created.", + }, + "task": { + "type": "string", + "description": "The task for this browser agent.", + }, + "url": { + "type": "string", + "description": "Optional starting URL.", + }, + }, + "required": ["task"], + }, + }, + }, + "required": ["tasks"], + }, + }, +] + + +def send_response(id_, result=None, error=None): + msg = {"jsonrpc": "2.0", "id": id_} + if error is not None: + msg["error"] = error + else: + msg["result"] = result + sys.stdout.write(json.dumps(msg) + "\n") + sys.stdout.flush() + + +def call_backend(tasks: list[dict]) -> dict: + pre_selected = [bid.strip() for bid in PRE_SELECTED_BROWSER_IDS.split(",") if bid.strip()] + payload = json.dumps({ + "tasks": tasks, + "model": MODEL, + "dashboard_id": DASHBOARD_ID, + "pre_selected_browser_ids": pre_selected, + }).encode() + req = urllib.request.Request( + BACKEND_URL, + data=payload, + headers={"Content-Type": "application/json"}, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=300) as resp: + return json.loads(resp.read().decode()) + except urllib.error.HTTPError as e: + body = e.read().decode() if e.fp else str(e) + return {"error": f"HTTP {e.code}: {body}"} + except Exception as e: + return {"error": str(e)} + + +MAX_IMAGE_B64_BYTES = 400_000 + + +def compress_screenshot(b64_png: str) -> tuple[str, str] | None: + """Resize and re-encode as JPEG to stay under the stdio buffer limit.""" + if not HAS_PIL: + return None + try: + raw = base64.b64decode(b64_png) + img = Image.open(BytesIO(raw)) + max_width = 1024 + if img.width > max_width: + ratio = max_width / img.width + img = img.resize((max_width, int(img.height * ratio)), Image.LANCZOS) + buf = BytesIO() + img.convert("RGB").save(buf, format="JPEG", quality=45) + return base64.b64encode(buf.getvalue()).decode(), "image/jpeg" + except Exception: + return None + + +def format_result(result: dict) -> dict: + """Format a single browser agent result into MCP content blocks.""" + if "error" in result: + return {"content": [{"type": "text", "text": f"Error: {result['error']}"}], "isError": True} + + content = [] + + summary = result.get("summary", "Task completed.") + session_id = result.get("session_id", "") + browser_id = result.get("browser_id", "") + action_log = result.get("action_log", []) + + lines = [f"**Browser Agent Result** (browser: {browser_id}, session: {session_id})", ""] + lines.append(f"**Summary:** {summary}") + + if action_log: + lines.append("") + lines.append("**Actions taken:**") + for i, entry in enumerate(action_log, 1): + tool = entry.get("tool", "?") + inp = entry.get("input", {}) + ms = entry.get("elapsed_ms", 0) + brief = json.dumps(inp)[:120] + lines.append(f" {i}. {tool}({brief}) [{ms}ms]") + + content.append({"type": "text", "text": "\n".join(lines)}) + + screenshot = result.get("final_screenshot") + if screenshot: + image_data = screenshot + mime_type = "image/png" + + if len(image_data) > MAX_IMAGE_B64_BYTES: + compressed = compress_screenshot(image_data) + if compressed: + image_data, mime_type = compressed + + if len(image_data) <= MAX_IMAGE_B64_BYTES: + content.append({"type": "image", "data": image_data, "mimeType": mime_type}) + content.append({"type": "text", "text": "Final screenshot attached above."}) + else: + content.append({"type": "text", "text": "Final screenshot was too large to include."}) + + return {"content": content} + + +def format_batch_results(results: list[dict]) -> dict: + """Format multiple browser agent results.""" + if isinstance(results, dict) and "error" in results: + return {"content": [{"type": "text", "text": f"Error: {results['error']}"}], "isError": True} + + all_content = [] + for i, result in enumerate(results): + formatted = format_result(result) + if i > 0: + all_content.append({"type": "text", "text": f"\n---\n"}) + all_content.extend(formatted.get("content", [])) + + return {"content": all_content} + + +def handle_tool_call(tool_name: str, arguments: dict) -> dict: + if tool_name == "BrowserAgent": + task_def = { + "task": arguments.get("task", ""), + "browser_id": arguments.get("browser_id", ""), + "url": arguments.get("url", ""), + } + result = call_backend([task_def]) + if "error" in result: + return {"content": [{"type": "text", "text": f"Error: {result['error']}"}], "isError": True} + results = result.get("results", [result]) + if results: + return format_result(results[0]) + return {"content": [{"type": "text", "text": "No result returned."}], "isError": True} + + elif tool_name == "BrowserAgents": + tasks = arguments.get("tasks", []) + if not tasks: + return {"content": [{"type": "text", "text": "Error: tasks array is empty"}], "isError": True} + result = call_backend(tasks) + if "error" in result: + return {"content": [{"type": "text", "text": f"Error: {result['error']}"}], "isError": True} + results = result.get("results", []) + return format_batch_results(results) + + return {"content": [{"type": "text", "text": f"Unknown tool: {tool_name}"}], "isError": True} + + +def main(): + for line in sys.stdin: + line = line.strip() + if not line: + continue + try: + msg = json.loads(line) + except json.JSONDecodeError: + continue + + method = msg.get("method") + id_ = msg.get("id") + params = msg.get("params", {}) + + if method == "initialize": + send_response(id_, { + "protocolVersion": "2024-11-05", + "capabilities": {"tools": {}}, + "serverInfo": { + "name": "openswarm-browser-agent", + "version": "1.0.0", + }, + }) + elif method == "notifications/initialized": + pass + elif method == "tools/list": + send_response(id_, {"tools": TOOLS}) + elif method == "tools/call": + tool_name = params.get("name", "") + arguments = params.get("arguments", {}) + result = handle_tool_call(tool_name, arguments) + send_response(id_, result) + elif method == "ping": + send_response(id_, {}) + elif id_ is not None: + send_response(id_, error={"code": -32601, "message": f"Method not found: {method}"}) + + +if __name__ == "__main__": + main() diff --git a/backend/apps/agents/models.py b/backend/apps/agents/models.py index d17828a5..a5f03df3 100644 --- a/backend/apps/agents/models.py +++ b/backend/apps/agents/models.py @@ -71,3 +71,4 @@ class AgentSession(BaseModel): active_branch_id: str = "main" tool_group_meta: dict[str, "ToolGroupMeta"] = Field(default_factory=dict) dashboard_id: Optional[str] = None + browser_id: Optional[str] = None diff --git a/backend/main.py b/backend/main.py index 6cd1935f..d7f30cee 100644 --- a/backend/main.py +++ b/backend/main.py @@ -114,6 +114,36 @@ async def browser_command(request: Request): return JSONResponse(result) +@app.post("/api/browser-agent/run") +async def browser_agent_run(request: Request): + """Run one or more browser sub-agents in parallel. + Called by the browser_agent_mcp_server stdio subprocess.""" + from backend.apps.settings.settings import load_settings + from backend.apps.agents.browser_agent import run_browser_agents + + body = await request.json() + tasks = body.get("tasks", []) + model = body.get("model", "sonnet") + dashboard_id = body.get("dashboard_id", "") + pre_selected_browser_ids = body.get("pre_selected_browser_ids", []) + + if not tasks: + return JSONResponse({"error": "tasks array is required"}, status_code=400) + + settings = load_settings() + if not settings.anthropic_api_key: + return JSONResponse({"error": "Anthropic API key not configured"}, status_code=400) + + results = await run_browser_agents( + tasks=tasks, + model=model, + api_key=settings.anthropic_api_key, + dashboard_id=dashboard_id or None, + pre_selected_browser_ids=pre_selected_browser_ids, + ) + return JSONResponse({"results": results}) + + if __name__ == "__main__": import argparse import uvicorn diff --git a/frontend/src/app/components/useDomElementSelector.ts b/frontend/src/app/components/useDomElementSelector.ts index 79031c0d..b4535e19 100644 --- a/frontend/src/app/components/useDomElementSelector.ts +++ b/frontend/src/app/components/useDomElementSelector.ts @@ -5,6 +5,9 @@ const SELECT_ATTR = 'data-select-type'; const SELECT_ID_ATTR = 'data-select-id'; const SELECT_META_ATTR = 'data-select-meta'; +const DRAG_SELECT_TYPES = ['agent-card', 'view-card', 'browser-card'] as const; +const DRAG_SELECTOR = DRAG_SELECT_TYPES.map((t) => `[${SELECT_ATTR}="${t}"]`).join(','); + export interface OverlayState { visible: boolean; top: number; @@ -167,7 +170,7 @@ export function useDomElementSelector(): DomSelectorState { dragPreviewRafRef.current = requestAnimationFrame(() => { const b = dragBoundsRef.current; if (!b) return; - const allSelectables = document.querySelectorAll(`[${SELECT_ATTR}]`); + const allSelectables = document.querySelectorAll(DRAG_SELECTOR); const preview: DragPreviewElement[] = []; const seen = new Set(); const excId = excludeIdRef.current; @@ -240,6 +243,7 @@ export function useDomElementSelector(): DomSelectorState { const handleMouseDown = useCallback((e: MouseEvent) => { if (e.button !== 0) return; + if (e.metaKey || e.ctrlKey) return; const target = e.target as Element; // Only start drag on "empty" canvas areas (not on selectable elements) if (target && findSelectableAncestor(target, excludeIdRef.current)) return; @@ -258,7 +262,7 @@ export function useDomElementSelector(): DomSelectorState { bottom: Math.max(dragOriginRef.current.y, e.clientY), }; - const allSelectables = document.querySelectorAll(`[${SELECT_ATTR}]`); + const allSelectables = document.querySelectorAll(DRAG_SELECTOR); const processed = new Set(); const excId = excludeIdRef.current; @@ -326,12 +330,16 @@ export function useDomElementSelector(): DomSelectorState { return; } + const prevUserSelect = document.body.style.userSelect; + document.body.style.userSelect = 'none'; + document.addEventListener('mousemove', handleMouseMove, true); document.addEventListener('mousedown', handleMouseDown, true); document.addEventListener('mouseup', handleMouseUp, true); document.addEventListener('click', handleClick, true); return () => { + document.body.style.userSelect = prevUserSelect; document.removeEventListener('mousemove', handleMouseMove, true); document.removeEventListener('mousedown', handleMouseDown, true); document.removeEventListener('mouseup', handleMouseUp, true); diff --git a/frontend/src/app/pages/AgentChat/ChatInput.tsx b/frontend/src/app/pages/AgentChat/ChatInput.tsx index 42b93e30..ff028b5d 100644 --- a/frontend/src/app/pages/AgentChat/ChatInput.tsx +++ b/frontend/src/app/pages/AgentChat/ChatInput.tsx @@ -293,34 +293,12 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, if (el.semanticType === 'browser-card' && el.semanticData?.selectId) { const wv = getWebview(el.semanticData.selectId as string); - if (wv) { - const url = el.semanticData.url || wv.getURL(); - const title = el.semanticData.name || wv.getTitle(); - lines.push(`${i + 1}. [Browser Card] ${title}`); - lines.push(` ID: ${el.semanticData.selectId}`); - lines.push(` URL: ${url}`); - - try { - const nativeImage = await wv.capturePage(); - const dataUrl = nativeImage.toDataURL(); - const base64 = dataUrl.replace(/^data:image\/\w+;base64,/, ''); - allImages.push({ data: base64, media_type: 'image/png' }); - lines.push(` [Screenshot captured and attached as image]`); - } catch { /* screenshot unavailable */ } - - try { - const pageText: string = await wv.executeJavaScript( - 'document.body.innerText.substring(0, 15000)' - ); - if (pageText?.trim()) { - lines.push(` Page text content:\n ---\n${pageText.trim().split('\n').map(l => ' ' + l).join('\n')}\n ---`); - } - } catch { /* text extraction unavailable */ } - } else { - lines.push(`${i + 1}. [Browser Card] ${el.semanticLabel || ''}`); - if (el.semanticData.selectId) lines.push(` ID: ${el.semanticData.selectId}`); - if (el.semanticData.url) lines.push(` URL: ${el.semanticData.url}`); - } + const url = wv ? (el.semanticData.url || wv.getURL()) : (el.semanticData.url || ''); + const title = wv ? (el.semanticData.name || wv.getTitle()) : (el.semanticLabel || ''); + lines.push(`${i + 1}. [Browser Card] ${title}`); + lines.push(` browser_id: ${el.semanticData.selectId}`); + if (url) lines.push(` URL: ${url}`); + lines.push(` (Use BrowserAgent with this browser_id to interact with it)`); } else if (el.semanticType && el.semanticData) { const typeLabel = { 'agent-card': 'Agent Card', diff --git a/frontend/src/app/pages/Dashboard/AgentCard.tsx b/frontend/src/app/pages/Dashboard/AgentCard.tsx index 7b0ef75b..31c2e07f 100644 --- a/frontend/src/app/pages/Dashboard/AgentCard.tsx +++ b/frontend/src/app/pages/Dashboard/AgentCard.tsx @@ -481,12 +481,35 @@ const AgentCard: React.FC = ({ /> ))} + {/* Selection overlay – blocks content interaction while selected, enabling drag from anywhere */} + {isSelected && ( + { + if (justDraggedRef.current) return; + onCardSelect?.(session.id, 'agent', e.shiftKey); + }} + onDoubleClick={() => dispatch(toggleExpandSession(session.id))} + sx={{ + position: 'absolute', + inset: 0, + zIndex: 15, + cursor: isDragging ? 'grabbing' : 'grab', + touchAction: 'none', + }} + /> + )} + {/* Drag zone: header + metadata – entire region above separator is draggable */} { try { return JSON.parse(msg.content); } catch { return {}; } })() : msg.content; + const tool = content?.tool || content?.name || '?'; + const input = content?.input || {}; + let brief = ''; + switch (tool) { + case 'BrowserNavigate': brief = `Navigate → ${input.url || '...'}`; break; + case 'BrowserClick': brief = `Click ${input.selector || '...'}`; break; + case 'BrowserType': brief = `Type "${(input.text || '').slice(0, 30)}${(input.text || '').length > 30 ? '…' : ''}" into ${input.selector || '...'}`; break; + case 'BrowserScreenshot': brief = 'Screenshot'; break; + case 'BrowserGetText': brief = 'Read page text'; break; + case 'BrowserGetElements': brief = `Inspect elements${input.selector ? ` (${input.selector})` : ''}`; break; + case 'BrowserEvaluate': brief = `Evaluate JS`; break; + default: brief = tool; + } + return { type: 'action', text: brief }; + } + + if (msg.role === 'tool_result') { + return { type: 'result', text: '' }; + } + + return { type: 'skip', text: '' }; +} + +const BrowserAgentOverlay: React.FC = ({ session, browserWidth, browserHeight }) => { + const c = useClaudeTokens(); + const dispatch = useAppDispatch(); + const scrollRef = useRef(null); + const [expanded, setExpanded] = useState(false); + const [confirmStop, setConfirmStop] = useState(false); + const [fadeOut, setFadeOut] = useState(false); + const confirmTimer = useRef | null>(null); + const fadeTimer = useRef | null>(null); + + const isRunning = session.status === 'running'; + const isDone = session.status === 'completed' || session.status === 'error' || session.status === 'stopped'; + + useEffect(() => { + if (isDone) { + fadeTimer.current = setTimeout(() => setFadeOut(true), 2000); + } + return () => { if (fadeTimer.current) clearTimeout(fadeTimer.current); }; + }, [isDone]); + + useEffect(() => { + if (scrollRef.current) { + scrollRef.current.scrollTop = scrollRef.current.scrollHeight; + } + }, [session.messages.length, session.streamingMessage]); + + const handleStop = useCallback(() => { + if (!confirmStop) { + setConfirmStop(true); + confirmTimer.current = setTimeout(() => setConfirmStop(false), 3000); + return; + } + if (confirmTimer.current) clearTimeout(confirmTimer.current); + setConfirmStop(false); + dispatch(stopAgent({ sessionId: session.id })); + }, [confirmStop, dispatch, session.id]); + + useEffect(() => { + return () => { if (confirmTimer.current) clearTimeout(confirmTimer.current); }; + }, []); + + const accentColor = c.accent.primary; + + const entries = session.messages + .map(summarizeMessage) + .filter((e) => e.type !== 'skip' && e.type !== 'result'); + + const streamingMsg = session.streamingMessage; + if (streamingMsg && streamingMsg.role === 'assistant' && streamingMsg.content) { + entries.push({ type: 'thought', text: streamingMsg.content }); + } + + const collapsedW = Math.min(300, browserWidth - 24); + const collapsedH = Math.min(200, browserHeight - 24); + const expandedW = Math.min(Math.floor(browserWidth * 0.55), browserWidth - 24); + const expandedH = Math.min(Math.floor(browserHeight * 0.6), browserHeight - 24); + + const panelW = expanded ? expandedW : collapsedW; + const panelH = expanded ? expandedH : collapsedH; + + if (fadeOut) return null; + + return ( + e.stopPropagation()} + onClick={(e) => e.stopPropagation()} + sx={{ + position: 'absolute', + bottom: 12, + right: 12, + width: panelW, + height: panelH, + zIndex: 18, + borderRadius: '12px', + bgcolor: 'rgba(15, 15, 15, 0.88)', + backdropFilter: 'blur(16px)', + border: `1px solid ${accentColor}30`, + boxShadow: `0 4px 24px rgba(0,0,0,0.5), 0 0 0 1px rgba(255,255,255,0.05)`, + display: 'flex', + flexDirection: 'column', + overflow: 'hidden', + transition: 'width 0.25s ease, height 0.25s ease, opacity 0.4s ease', + opacity: isDone && !fadeOut ? 0.7 : 1, + animation: 'overlay-enter 0.3s ease-out', + '@keyframes overlay-enter': { + '0%': { opacity: 0, transform: 'translateY(8px) scale(0.95)' }, + '100%': { opacity: 1, transform: 'translateY(0) scale(1)' }, + }, + }} + > + {/* Header */} + + + + {isRunning && ( + + )} + + {isDone && session.status === 'completed' && ( + + )} + {isDone && session.status === 'error' && ( + + )} + + + {isDone + ? session.status === 'completed' ? 'Done' : session.status === 'error' ? 'Error' : 'Stopped' + : 'Browser Agent'} + + + + setExpanded((e) => !e)} + sx={{ + color: 'rgba(255,255,255,0.5)', + p: 0.3, + '&:hover': { color: 'rgba(255,255,255,0.8)' }, + }} + > + {expanded + ? + : + } + + + + {isRunning && ( + + + + + + )} + + + {/* Body — scrollable action log */} + + {entries.length === 0 && isRunning && ( + + Starting... + + )} + + {entries.map((entry, i) => ( + + {entry.type === 'thought' ? ( + <> + + + {entry.text} + + + ) : ( + <> + + + {entry.text} + + + )} + + ))} + + + ); +}; + +export default React.memo(BrowserAgentOverlay); diff --git a/frontend/src/app/pages/Dashboard/BrowserCard.tsx b/frontend/src/app/pages/Dashboard/BrowserCard.tsx index 1eb5aa11..d7772294 100644 --- a/frontend/src/app/pages/Dashboard/BrowserCard.tsx +++ b/frontend/src/app/pages/Dashboard/BrowserCard.tsx @@ -39,6 +39,8 @@ import { import { useBrowserActivity } from '@/shared/useBrowserActivity'; import { getActionLabel } from '@/shared/browserCommandHandler'; import { resolveInput, isGoogleSearch } from '@/shared/resolveUrl'; +import BrowserAgentOverlay from './BrowserAgentOverlay'; +import { useElementSelection } from '@/app/components/ElementSelectionContext'; type ResizeDir = 'n' | 's' | 'e' | 'w' | 'ne' | 'nw' | 'se' | 'sw'; @@ -99,6 +101,16 @@ const BrowserCard: React.FC = ({ const c = useClaudeTokens(); const dispatch = useAppDispatch(); const browserHomepage = useAppSelector((state) => state.settings.data.browser_homepage); + const elementSelectionCtx = useElementSelection(); + const isElementSelectMode = elementSelectionCtx?.selectMode ?? false; + + const browserAgentSession = useAppSelector((state) => { + const sessions = state.agents.sessions; + return Object.values(sessions).find( + (s) => s.browser_id === browserId && s.mode === 'browser-agent' + && (s.status === 'running' || s.status === 'completed' || s.status === 'error'), + ) ?? null; + }); const activity = useBrowserActivity(browserId); const agentActive = activity.active; @@ -576,6 +588,26 @@ const BrowserCard: React.FC = ({ }), }} > + {/* Selection overlay – blocks content interaction while selected, enabling drag from anywhere */} + {isSelected && ( + { + if (justDraggedRef.current) return; + onCardSelect?.(browserId, 'browser', e.shiftKey); + }} + sx={{ + position: 'absolute', + inset: 0, + zIndex: 15, + cursor: isDragging ? 'grabbing' : 'grab', + touchAction: 'none', + }} + /> + )} + {/* Rotating gradient border glow for element selection / streaming */} {showGlow && !agentActive && ( = ({ onPointerMove={handleDragPointerMove} onPointerUp={handleDragPointerUp} sx={{ + position: 'relative', + zIndex: 16, display: 'flex', alignItems: 'stretch', bgcolor: agentActive ? `${accentColor}0a` : c.bg.secondary, @@ -940,6 +974,9 @@ const BrowserCard: React.FC = ({ {/* ====== Browser body — multiple webviews stacked ====== */} + {isElementSelectMode && ( + + )} {isElectron ? ( tabs.map((tab) => ( = ({ )} {/* ===== Frosted glass overlay ===== */} - {agentActive && ( + {agentActive && !browserAgentSession && ( = ({ )} + + {/* ===== Browser Agent Overlay ===== */} + {browserAgentSession && ( + + )} {/* Resize handles */} diff --git a/frontend/src/app/pages/Dashboard/Dashboard.tsx b/frontend/src/app/pages/Dashboard/Dashboard.tsx index 8a5521a8..2aa9393b 100644 --- a/frontend/src/app/pages/Dashboard/Dashboard.tsx +++ b/frontend/src/app/pages/Dashboard/Dashboard.tsx @@ -42,7 +42,7 @@ import { useDashboardSelection } from './useDashboardSelection'; import type { CardType } from './useDashboardSelection'; import { useClaudeTokens } from '@/shared/styles/ThemeContext'; import type { ContextPath } from '@/app/components/DirectoryBrowser'; -import { ElementSelectionProvider } from '@/app/components/ElementSelectionContext'; +import { ElementSelectionProvider, useElementSelection } from '@/app/components/ElementSelectionContext'; import { useDomElementSelector } from '@/app/components/useDomElementSelector'; import SelectionOverlay from '@/app/components/SelectionOverlay'; @@ -66,6 +66,8 @@ const DashboardInner: React.FC = () => { const c = useClaudeTokens(); const dispatch = useAppDispatch(); const { id: dashboardId } = useParams<{ id: string }>(); + const elementSelectionCtx = useElementSelection(); + const isElementSelectMode = elementSelectionCtx?.selectMode ?? false; const dashboardName = useAppSelector((state) => dashboardId ? state.dashboards.items[dashboardId]?.name : undefined, ); @@ -158,12 +160,21 @@ const DashboardInner: React.FC = () => { if (e.button !== 0) return; if (isCardTarget(e.target, e.currentTarget)) return; + if (isElementSelectMode) { + // Cmd/Ctrl held → allow panning even in element select mode + if (e.metaKey || e.ctrlKey) { + canvas.handlers.onMouseDown(e); + } + return; + } + if (e.metaKey || e.ctrlKey || canvas.spaceHeld) { selection.handleCanvasMouseDown(e.nativeEvent); } else { + selection.deselectAll(); canvas.handlers.onMouseDown(e); } - }, [canvas.handlers, canvas.spaceHeld, selection]); + }, [canvas.handlers, canvas.spaceHeld, selection, isElementSelectMode]); const handleViewportMouseMove = useCallback((e: React.MouseEvent) => { canvas.handlers.onMouseMove(e); @@ -252,7 +263,7 @@ const DashboardInner: React.FC = () => { useEffect(() => { if (!layoutInitialized) return; const dashboardSessionIds = Object.values(sessions) - .filter((s) => s.dashboard_id === dashboardId) + .filter((s) => s.dashboard_id === dashboardId && s.mode !== 'browser-agent') .map((s) => s.id); const liveIds = dashboardSessionIds.sort().join(','); if (liveIds === prevSessionIdsRef.current) return; diff --git a/frontend/src/app/pages/Dashboard/DashboardViewCard.tsx b/frontend/src/app/pages/Dashboard/DashboardViewCard.tsx index b906d30a..19c7fd0b 100644 --- a/frontend/src/app/pages/Dashboard/DashboardViewCard.tsx +++ b/frontend/src/app/pages/Dashboard/DashboardViewCard.tsx @@ -304,12 +304,34 @@ const DashboardViewCard: React.FC = ({ }), }} > + {/* Selection overlay – blocks content interaction while selected, enabling drag from anywhere */} + {isSelected && ( + { + if (justDraggedRef.current) return; + onCardSelect?.(output.id, 'view', e.shiftKey); + }} + sx={{ + position: 'absolute', + inset: 0, + zIndex: 15, + cursor: isDragging ? 'grabbing' : 'grab', + touchAction: 'none', + }} + /> + )} + {/* Header */} { Prepended to every agent session before mode-specific instructions. Modes can override with their own. - setForm({ ...form, default_system_prompt: e.target.value || null })} + multiline + minRows={3} + maxRows={8} + fullWidth + size="small" sx={{ - p: 1.5, - borderRadius: `${c.radius.sm}px`, - border: `1px solid ${c.border.subtle}`, - bgcolor: c.bg.secondary, - fontFamily: c.font.mono, - fontSize: '0.8rem', - color: c.text.secondary, - lineHeight: 1.6, - whiteSpace: 'pre-wrap', - wordBreak: 'break-word', - maxHeight: 200, - overflow: 'auto', - '&::-webkit-scrollbar': { width: 5 }, - '&::-webkit-scrollbar-track': { background: 'transparent' }, - '&::-webkit-scrollbar-thumb': { background: c.border.medium, borderRadius: 3, '&:hover': { background: c.border.strong } }, - scrollbarWidth: 'thin', - scrollbarColor: `${c.border.medium} transparent`, + '& .MuiOutlinedInput-root': { + fontFamily: c.font.mono, + fontSize: '0.8rem', + lineHeight: 1.6, + color: c.text.secondary, + }, }} - > - {form.default_system_prompt || DEFAULT_SYSTEM_PROMPT} - + /> diff --git a/frontend/src/shared/state/agentsSlice.ts b/frontend/src/shared/state/agentsSlice.ts index 2186a419..4fb21107 100644 --- a/frontend/src/shared/state/agentsSlice.ts +++ b/frontend/src/shared/state/agentsSlice.ts @@ -68,6 +68,7 @@ export interface AgentSession { target_directory?: string | null; tool_group_meta: Record; dashboard_id?: string; + browser_id?: string | null; } export interface AgentConfig { diff --git a/frontend/src/shared/state/dashboardLayoutSlice.ts b/frontend/src/shared/state/dashboardLayoutSlice.ts index 8ac1795f..721aef6c 100644 --- a/frontend/src/shared/state/dashboardLayoutSlice.ts +++ b/frontend/src/shared/state/dashboardLayoutSlice.ts @@ -350,6 +350,16 @@ const dashboardLayoutSlice = createSlice({ }; }, + addBrowserCardFromBackend(state, action: PayloadAction) { + const card = action.payload; + if (state.browserCards[card.browser_id]) return; + state.browserCards[card.browser_id] = { + ...card, + width: card.width || DEFAULT_BROWSER_CARD_W, + height: card.height || DEFAULT_BROWSER_CARD_H, + }; + }, + setBrowserCardPosition( state, action: PayloadAction<{ browserId: string; x: number; y: number }> @@ -596,6 +606,7 @@ export const { setViewCardSize, removeViewCard, addBrowserCard, + addBrowserCardFromBackend, setBrowserCardPosition, setBrowserCardSize, removeBrowserCard, diff --git a/frontend/src/shared/state/settingsSlice.ts b/frontend/src/shared/state/settingsSlice.ts index baeec919..58961793 100644 --- a/frontend/src/shared/state/settingsSlice.ts +++ b/frontend/src/shared/state/settingsSlice.ts @@ -8,7 +8,8 @@ export const DEFAULT_SYSTEM_PROMPT = `When asking the user questions, use the AskUserQuestion tool.\n\n` + `You are an unstopable Agent that does whatever is needed to achieve the task. ` + `You are particularly gifted at coding, so when needed, transpose ordinary tasks into coding tasks.\n\n` + - `If a Browser is selected, prioritize this over other tools when it makes sense (so the user also has observability).`; + `If a Browser is selected, prioritize this over other tools when it makes sense (so the user also has observability).\n\n` + + `If multiple Browsers are selected, parallelize the tasks across them.`; export interface AppSettings { default_system_prompt: string | null; diff --git a/frontend/src/shared/ws/WebSocketManager.ts b/frontend/src/shared/ws/WebSocketManager.ts index 6999f43f..71b5dce4 100644 --- a/frontend/src/shared/ws/WebSocketManager.ts +++ b/frontend/src/shared/ws/WebSocketManager.ts @@ -15,6 +15,7 @@ import { setActiveBranch, closeSessionFromWs, } from '../state/agentsSlice'; +import { addBrowserCardFromBackend } from '../state/dashboardLayoutSlice'; type WSEvent = { event: string; @@ -230,6 +231,12 @@ class WebSocketManager { })); } break; + + case 'dashboard:browser_card_added': + if (data.browser_card) { + store.dispatch(addBrowserCardFromBackend(data.browser_card)); + } + break; } // Notify any custom listeners