diff --git a/backend/apps/agents/browser_agent.py b/backend/apps/agents/browser_agent.py index 728a07f6..a9463f4b 100644 --- a/backend/apps/agents/browser_agent.py +++ b/backend/apps/agents/browser_agent.py @@ -15,735 +15,35 @@ from uuid import uuid4 import anthropic +from backend.apps.agents import browser_history +from backend.apps.agents.browser_history import ( + _MAX_HISTORY_MESSAGES, + _trim_history_by_turns, + _validate_message_pairing, + clear_browser_history, +) +from backend.apps.agents.browser_loop import ( + _LOOP_DETECTION_EXCLUDED_TOOLS, + _LOOP_HARD_CAP, + _LOOP_WARNING_TEXT, + _LOOP_WINDOW_SIZE, + _detect_loop, + _hash_tool_call, +) +from backend.apps.agents.browser_schema import ( + _ACTION_TOOLS_REQUIRING_REPORT, + ACTION_MAP, + BROWSER_TOOLS_SCHEMA, + MAX_TURNS, + MODEL_MAP, + SYSTEM_PROMPT, +) from backend.apps.agents.models import AgentSession, ApprovalRequest, Message from backend.apps.agents.ws_manager import ws_manager from backend.apps.tools_lib.tools_lib import load_builtin_permissions logger = logging.getLogger(__name__) -MODEL_MAP = { - "sonnet": "claude-sonnet-4-6", - "opus": "claude-opus-4-6", - "haiku": "claude-haiku-4-5-20251001", -} - -# Cache of conversation history per browser_id so successive BrowserAgent -# calls on the same browser can resume rather than restart from scratch. -# Without this every "swipe right" / "swipe left" call has to take a new -# screenshot and re-orient itself, costing 30-60s per action. -_browser_history: dict[str, list[dict]] = {} -# Cap history to prevent unbounded growth on long-lived browsers. -_MAX_HISTORY_MESSAGES = 30 - - -def clear_browser_history(browser_id: str) -> None: - """Drop cached conversation history for a browser (e.g. when it's closed).""" - _browser_history.pop(browser_id, None) - - -# --------------------------------------------------------------------------- -# Loop detection -# -# Tracks recent state-mutating tool calls in a sliding window. If the model -# repeats the same (tool, input) with the same result several times, we -# inject an is_error message in the next tool_result to force a strategy -# change. This prevents the model from burning the entire turn budget on -# a failing approach. -# --------------------------------------------------------------------------- - -# Tools that are read-only / idempotent and should NOT count toward loop -# detection. Repeating these is normal (scrolling through a feed, taking -# successive screenshots, polling for an element to appear). -_LOOP_DETECTION_EXCLUDED_TOOLS = { - "BrowserScreenshot", - "BrowserGetText", - "BrowserGetElements", - "BrowserListInteractives", # Phase 3 - "BrowserWait", - "ReportProgress", # Phase 2 - "RequestHumanIntervention", -} - -_LOOP_WINDOW_SIZE = 5 -_LOOP_REPEAT_THRESHOLD = 3 -_LOOP_HARD_CAP = 5 - - -def _hash_tool_call(tool_name: str, tool_input: dict, result: dict) -> tuple[str, str, str]: - """Build a stable hash key for a tool call, including its result. - - Including the result hash means that legitimate progress (same input, - different output; e.g. BrowserScroll on a long feed) does NOT count - as a loop. Only same-input + same-output is treated as stuck. - """ - try: - input_key = json.dumps(tool_input, sort_keys=True, default=str) - except Exception: - input_key = repr(tool_input) - try: - # Truncate the result hash to avoid huge image blobs in the key - result_key = json.dumps(result, sort_keys=True, default=str)[:300] - except Exception: - result_key = repr(result)[:300] - return (tool_name, input_key, result_key) - - -def _detect_loop( - recent_calls: list[tuple[str, str, str]], - new_call: tuple[str, str, str], -) -> bool: - """Return True if `new_call` constitutes a loop given recent history. - - A loop is when the same (tool, input, result) has appeared at least - `_LOOP_REPEAT_THRESHOLD` times within the last `_LOOP_WINDOW_SIZE` - state-mutating calls (the new call counts as one of those occurrences). - """ - if new_call[0] in _LOOP_DETECTION_EXCLUDED_TOOLS: - return False - window = recent_calls[-(_LOOP_WINDOW_SIZE - 1):] + [new_call] - matches = sum(1 for c in window if c == new_call) - return matches >= _LOOP_REPEAT_THRESHOLD - - -_LOOP_WARNING_TEXT = ( - "LOOP DETECTED: You have called this tool with these exact parameters and " - "gotten the same result {count} times in a row. STOP retrying this approach " - ", it is not working. Try a fundamentally different strategy: " - "(1) check the page state with BrowserScreenshot or BrowserGetText, " - "(2) try a different selector or a different tool, " - "(3) use BrowserPressKey for keyboard shortcuts if the site supports them, " - "or (4) call RequestHumanIntervention if you genuinely cannot proceed." -) - - -def _validate_message_pairing(messages: list[dict]) -> bool: - """Verify every tool_result references a tool_use_id from a prior assistant - message in the same list. Returns False if there's an orphan, which means - the cached history would 400 if sent to the API. - - This is the last line of defense against cache corruption; if it ever - returns False on a resume, we drop the cache and start fresh rather than - crash on the next API call. - """ - declared_tool_use_ids: set[str] = set() - for msg in messages: - role = msg.get("role") - content = msg.get("content") - if role == "assistant" and isinstance(content, list): - for block in content: - if isinstance(block, dict) and block.get("type") == "tool_use": - tu_id = block.get("id") - if tu_id: - declared_tool_use_ids.add(tu_id) - elif role == "user" and isinstance(content, list): - for block in content: - if isinstance(block, dict) and block.get("type") == "tool_result": - tr_id = block.get("tool_use_id") - if tr_id and tr_id not in declared_tool_use_ids: - return False - return True - - -def _is_fresh_user_message(msg: dict) -> bool: - """A 'fresh' user message starts a new turn; string content or a list - that contains no tool_result blocks. These are the only safe cut points - because they don't reference any prior assistant tool_use blocks.""" - if msg.get("role") != "user": - return False - content = msg.get("content") - if isinstance(content, str): - return True - if isinstance(content, list) and not any( - isinstance(c, dict) and c.get("type") == "tool_result" for c in content - ): - return True - return False - - -def _summarize_messages(messages: list[dict]) -> str: - """Build a programmatic summary of older browser-agent messages. - - Extracts the original user task, a count of tool calls by name with their - key parameters, the last few ReportProgress brain states, and the most - recent assistant text. No LLM call required; this is purely structural - extraction from the existing message history. - """ - if not messages: - return "" - - # Find the original user task (first user-text message) - initial_task = "" - for msg in messages: - if msg.get("role") == "user": - content = msg.get("content") - if isinstance(content, str) and content.strip(): - initial_task = content.strip()[:300] - break - - # Count tool calls by name with key params - tool_call_summary: dict[str, list[str]] = {} - brain_states: list[str] = [] - last_assistant_text = "" - - for msg in messages: - if msg.get("role") != "assistant": - continue - content = msg.get("content") - if not isinstance(content, list): - continue - for block in content: - if not isinstance(block, dict): - continue - btype = block.get("type") - if btype == "tool_use": - name = block.get("name", "unknown") - inp = block.get("input") or {} - if name == "ReportProgress": - # Capture the brain state for inline summary - brain_states.append( - f" • {inp.get('next_goal', '')[:120]}" - ) - continue - # Compact one-line description with key params - key_param = "" - for k in ("index", "key", "url", "selector", "direction", "text"): - if k in inp: - v = str(inp[k])[:40] - key_param = f"{k}={v}" - break - desc = f"{name}({key_param})" if key_param else name - tool_call_summary.setdefault(name, []).append(desc) - elif btype == "text": - txt = block.get("text", "").strip() - if txt: - last_assistant_text = txt - - # Build the summary text - parts = ["[Summary of earlier browser-agent activity]"] - if initial_task: - parts.append(f'Original task: "{initial_task}"') - if tool_call_summary: - total = sum(len(v) for v in tool_call_summary.values()) - parts.append(f"Actions taken ({total} total):") - # Show count + a couple of representative examples per tool - for name in sorted(tool_call_summary.keys()): - calls = tool_call_summary[name] - count = len(calls) - sample = calls[-1] # most recent example - if count == 1: - parts.append(f" - {sample}") - else: - parts.append(f" - {sample} (×{count})") - if brain_states: - parts.append("Recent intents:") - parts.extend(brain_states[-5:]) # last 5 brain states - if last_assistant_text: - snippet = last_assistant_text[:400] - parts.append(f"Last update from assistant: {snippet}") - parts.append( - "(Earlier turn-by-turn details have been compacted to keep the " - "context window manageable. Continue from where you left off.)" - ) - return "\n".join(parts) - - -def _trim_history_by_turns(messages: list[dict], max_messages: int) -> list[dict]: - """Compact message history when it exceeds max_messages. - - The Anthropic API requires every `tool_result` block to reference a - `tool_use_id` from a previous assistant message. Naive slicing can drop - a tool_use while keeping its tool_result, causing 400 errors. This - function avoids that by: - - 1. Walking forward to find a clean turn boundary (a fresh user-text - message that starts a new turn; no tool_result content). - 2. Summarizing everything BEFORE that boundary into a single user-text - message and prepending it to the kept tail. - 3. If no clean boundary exists at all, returning the original history - unchanged. Better to temporarily exceed the cap than to corrupt the - conversation and 400 every subsequent request. - - The summary is built programmatically (no LLM call) from the message - structure: original task, tool call counts, recent ReportProgress brain - states, and last assistant text. - """ - if len(messages) <= max_messages: - return list(messages) - - target_tail_size = max_messages - 1 # leave room for the summary message - cut_index: int | None = None - - # First pass: walk forward looking for the EARLIEST clean cut point that - # gets us under the cap. This preserves the most recent detail. - for i in range(1, len(messages)): - if not _is_fresh_user_message(messages[i]): - continue - if len(messages) - i <= target_tail_size: - cut_index = i - break - - # Second pass: if no cut point gets us under the cap (e.g. the current - # turn alone is bigger than max_messages), use the LATEST clean cut point - # available. The tail will still exceed the cap, but it's the smallest - # safe history we can produce; and any compaction is better than none. - if cut_index is None: - for i in range(len(messages) - 1, 0, -1): - if _is_fresh_user_message(messages[i]): - cut_index = i - break - - if cut_index is None: - # No clean cut anywhere in the history. Return original; better to - # exceed the cap than to corrupt the conversation. - return list(messages) - - # Compact: summarize messages[0..cut_index-1], prepend as a single - # user-text message, then keep messages[cut_index..end] verbatim. - summary_text = _summarize_messages(messages[:cut_index]) - summary_msg = {"role": "user", "content": summary_text} - return [summary_msg] + list(messages[cut_index:]) - -BROWSER_TOOLS_SCHEMA = [ - { - "name": "ReportProgress", - "description": ( - "Record your assessment of the previous action and your plan for the " - "next one. You MUST call this BEFORE any browser action tools in every " - "turn (after the very first turn). This is how you reflect on what just " - "happened, track what you've learned about this site, and articulate what " - "you're trying to do next. Skipping it is not allowed and will be rejected." - ), - "input_schema": { - "type": "object", - "properties": { - "evaluation_previous": { - "type": "string", - "description": ( - "What did the previous action(s) accomplish? Did they succeed? " - "If not, why? Be specific about what changed on the page." - ), - }, - "working_memory": { - "type": "string", - "description": ( - "Short notes about what you've learned about this site so far; " - "selectors that work, keyboard shortcuts, layout quirks, what " - "you've tried that failed. Carry this forward across turns." - ), - }, - "next_goal": { - "type": "string", - "description": ( - "What you're trying to achieve with the action(s) you're about " - "to take next. Be concrete." - ), - }, - }, - "required": ["evaluation_previous", "working_memory", "next_goal"], - }, - }, - { - "name": "BrowserScreenshot", - "description": ( - "Capture a screenshot of the browser page. Returns the screenshot as a " - "base64-encoded PNG image. Use this to see what is currently displayed." - ), - "input_schema": { - "type": "object", - "properties": {}, - "required": [], - }, - }, - { - "name": "BrowserGetText", - "description": ( - "Get the visible text content of the browser page. Returns up to 15000 characters." - ), - "input_schema": { - "type": "object", - "properties": {}, - "required": [], - }, - }, - { - "name": "BrowserNavigate", - "description": "Navigate the browser to a URL.", - "input_schema": { - "type": "object", - "properties": { - "url": {"type": "string", "description": "The URL to navigate to."}, - }, - "required": ["url"], - }, - }, - { - "name": "BrowserClick", - "description": "Click an element identified by a CSS selector. Use BrowserGetElements first to discover valid selectors.", - "input_schema": { - "type": "object", - "properties": { - "selector": {"type": "string", "description": "CSS selector of the element to click."}, - }, - "required": ["selector"], - }, - }, - { - "name": "BrowserType", - "description": "Type text into an input element. Clears existing value first.", - "input_schema": { - "type": "object", - "properties": { - "selector": {"type": "string", "description": "CSS selector of the input element."}, - "text": {"type": "string", "description": "The text to type."}, - }, - "required": ["selector", "text"], - }, - }, - { - "name": "BrowserEvaluate", - "description": "Evaluate a JavaScript expression in the browser page and return the result.", - "input_schema": { - "type": "object", - "properties": { - "expression": {"type": "string", "description": "JavaScript expression to evaluate."}, - }, - "required": ["expression"], - }, - }, - { - "name": "BrowserGetElements", - "description": ( - "Get a list of interactive elements on the page with CSS selectors. " - "Call this BEFORE clicking or typing so you know which selectors are valid." - ), - "input_schema": { - "type": "object", - "properties": { - "selector": { - "type": "string", - "description": "Optional CSS selector to scope the search (e.g. 'form', '#main'). Defaults to 'body'.", - }, - }, - "required": [], - }, - }, - { - "name": "BrowserScroll", - "description": ( - "Scroll the page up or down. Automatically finds the correct scrollable " - "container (works on SPAs like Notion, Gmail, etc. that use nested scroll " - "containers instead of window-level scrolling). Returns scroll position info " - "including whether top/bottom has been reached." - ), - "input_schema": { - "type": "object", - "properties": { - "direction": { - "type": "string", - "enum": ["up", "down"], - "description": "Scroll direction. Defaults to 'down'.", - }, - "amount": { - "type": "number", - "description": "Pixels to scroll. Defaults to 500.", - }, - }, - "required": [], - }, - }, - { - "name": "BrowserListInteractives", - "description": ( - "Get a NUMBERED LIST of interactive elements on the page using the " - "browser's accessibility tree. Returns elements like [1]