[eric] split: carve browser_agent schema/history/loop into modules

This commit is contained in:
ciregenz
2026-05-23 02:48:05 -07:00
parent a8843d58c8
commit 567ed5236c
4 changed files with 763 additions and 726 deletions
+26 -726
View File
@@ -15,735 +15,35 @@ from uuid import uuid4
import anthropic
from backend.apps.agents import browser_history
from backend.apps.agents.browser_history import (
_MAX_HISTORY_MESSAGES,
_trim_history_by_turns,
_validate_message_pairing,
clear_browser_history,
)
from backend.apps.agents.browser_loop import (
_LOOP_DETECTION_EXCLUDED_TOOLS,
_LOOP_HARD_CAP,
_LOOP_WARNING_TEXT,
_LOOP_WINDOW_SIZE,
_detect_loop,
_hash_tool_call,
)
from backend.apps.agents.browser_schema import (
_ACTION_TOOLS_REQUIRING_REPORT,
ACTION_MAP,
BROWSER_TOOLS_SCHEMA,
MAX_TURNS,
MODEL_MAP,
SYSTEM_PROMPT,
)
from backend.apps.agents.models import AgentSession, ApprovalRequest, Message
from backend.apps.agents.ws_manager import ws_manager
from backend.apps.tools_lib.tools_lib import load_builtin_permissions
logger = logging.getLogger(__name__)
MODEL_MAP = {
"sonnet": "claude-sonnet-4-6",
"opus": "claude-opus-4-6",
"haiku": "claude-haiku-4-5-20251001",
}
# Cache of conversation history per browser_id so successive BrowserAgent
# calls on the same browser can resume rather than restart from scratch.
# Without this every "swipe right" / "swipe left" call has to take a new
# screenshot and re-orient itself, costing 30-60s per action.
_browser_history: dict[str, list[dict]] = {}
# Cap history to prevent unbounded growth on long-lived browsers.
_MAX_HISTORY_MESSAGES = 30
def clear_browser_history(browser_id: str) -> None:
"""Drop cached conversation history for a browser (e.g. when it's closed)."""
_browser_history.pop(browser_id, None)
# ---------------------------------------------------------------------------
# Loop detection
#
# Tracks recent state-mutating tool calls in a sliding window. If the model
# repeats the same (tool, input) with the same result several times, we
# inject an is_error message in the next tool_result to force a strategy
# change. This prevents the model from burning the entire turn budget on
# a failing approach.
# ---------------------------------------------------------------------------
# Tools that are read-only / idempotent and should NOT count toward loop
# detection. Repeating these is normal (scrolling through a feed, taking
# successive screenshots, polling for an element to appear).
_LOOP_DETECTION_EXCLUDED_TOOLS = {
"BrowserScreenshot",
"BrowserGetText",
"BrowserGetElements",
"BrowserListInteractives", # Phase 3
"BrowserWait",
"ReportProgress", # Phase 2
"RequestHumanIntervention",
}
_LOOP_WINDOW_SIZE = 5
_LOOP_REPEAT_THRESHOLD = 3
_LOOP_HARD_CAP = 5
def _hash_tool_call(tool_name: str, tool_input: dict, result: dict) -> tuple[str, str, str]:
"""Build a stable hash key for a tool call, including its result.
Including the result hash means that legitimate progress (same input,
different output; e.g. BrowserScroll on a long feed) does NOT count
as a loop. Only same-input + same-output is treated as stuck.
"""
try:
input_key = json.dumps(tool_input, sort_keys=True, default=str)
except Exception:
input_key = repr(tool_input)
try:
# Truncate the result hash to avoid huge image blobs in the key
result_key = json.dumps(result, sort_keys=True, default=str)[:300]
except Exception:
result_key = repr(result)[:300]
return (tool_name, input_key, result_key)
def _detect_loop(
recent_calls: list[tuple[str, str, str]],
new_call: tuple[str, str, str],
) -> bool:
"""Return True if `new_call` constitutes a loop given recent history.
A loop is when the same (tool, input, result) has appeared at least
`_LOOP_REPEAT_THRESHOLD` times within the last `_LOOP_WINDOW_SIZE`
state-mutating calls (the new call counts as one of those occurrences).
"""
if new_call[0] in _LOOP_DETECTION_EXCLUDED_TOOLS:
return False
window = recent_calls[-(_LOOP_WINDOW_SIZE - 1):] + [new_call]
matches = sum(1 for c in window if c == new_call)
return matches >= _LOOP_REPEAT_THRESHOLD
_LOOP_WARNING_TEXT = (
"LOOP DETECTED: You have called this tool with these exact parameters and "
"gotten the same result {count} times in a row. STOP retrying this approach "
", it is not working. Try a fundamentally different strategy: "
"(1) check the page state with BrowserScreenshot or BrowserGetText, "
"(2) try a different selector or a different tool, "
"(3) use BrowserPressKey for keyboard shortcuts if the site supports them, "
"or (4) call RequestHumanIntervention if you genuinely cannot proceed."
)
def _validate_message_pairing(messages: list[dict]) -> bool:
"""Verify every tool_result references a tool_use_id from a prior assistant
message in the same list. Returns False if there's an orphan, which means
the cached history would 400 if sent to the API.
This is the last line of defense against cache corruption; if it ever
returns False on a resume, we drop the cache and start fresh rather than
crash on the next API call.
"""
declared_tool_use_ids: set[str] = set()
for msg in messages:
role = msg.get("role")
content = msg.get("content")
if role == "assistant" and isinstance(content, list):
for block in content:
if isinstance(block, dict) and block.get("type") == "tool_use":
tu_id = block.get("id")
if tu_id:
declared_tool_use_ids.add(tu_id)
elif role == "user" and isinstance(content, list):
for block in content:
if isinstance(block, dict) and block.get("type") == "tool_result":
tr_id = block.get("tool_use_id")
if tr_id and tr_id not in declared_tool_use_ids:
return False
return True
def _is_fresh_user_message(msg: dict) -> bool:
"""A 'fresh' user message starts a new turn; string content or a list
that contains no tool_result blocks. These are the only safe cut points
because they don't reference any prior assistant tool_use blocks."""
if msg.get("role") != "user":
return False
content = msg.get("content")
if isinstance(content, str):
return True
if isinstance(content, list) and not any(
isinstance(c, dict) and c.get("type") == "tool_result" for c in content
):
return True
return False
def _summarize_messages(messages: list[dict]) -> str:
"""Build a programmatic summary of older browser-agent messages.
Extracts the original user task, a count of tool calls by name with their
key parameters, the last few ReportProgress brain states, and the most
recent assistant text. No LLM call required; this is purely structural
extraction from the existing message history.
"""
if not messages:
return ""
# Find the original user task (first user-text message)
initial_task = ""
for msg in messages:
if msg.get("role") == "user":
content = msg.get("content")
if isinstance(content, str) and content.strip():
initial_task = content.strip()[:300]
break
# Count tool calls by name with key params
tool_call_summary: dict[str, list[str]] = {}
brain_states: list[str] = []
last_assistant_text = ""
for msg in messages:
if msg.get("role") != "assistant":
continue
content = msg.get("content")
if not isinstance(content, list):
continue
for block in content:
if not isinstance(block, dict):
continue
btype = block.get("type")
if btype == "tool_use":
name = block.get("name", "unknown")
inp = block.get("input") or {}
if name == "ReportProgress":
# Capture the brain state for inline summary
brain_states.append(
f"{inp.get('next_goal', '')[:120]}"
)
continue
# Compact one-line description with key params
key_param = ""
for k in ("index", "key", "url", "selector", "direction", "text"):
if k in inp:
v = str(inp[k])[:40]
key_param = f"{k}={v}"
break
desc = f"{name}({key_param})" if key_param else name
tool_call_summary.setdefault(name, []).append(desc)
elif btype == "text":
txt = block.get("text", "").strip()
if txt:
last_assistant_text = txt
# Build the summary text
parts = ["[Summary of earlier browser-agent activity]"]
if initial_task:
parts.append(f'Original task: "{initial_task}"')
if tool_call_summary:
total = sum(len(v) for v in tool_call_summary.values())
parts.append(f"Actions taken ({total} total):")
# Show count + a couple of representative examples per tool
for name in sorted(tool_call_summary.keys()):
calls = tool_call_summary[name]
count = len(calls)
sample = calls[-1] # most recent example
if count == 1:
parts.append(f" - {sample}")
else:
parts.append(f" - {sample} (×{count})")
if brain_states:
parts.append("Recent intents:")
parts.extend(brain_states[-5:]) # last 5 brain states
if last_assistant_text:
snippet = last_assistant_text[:400]
parts.append(f"Last update from assistant: {snippet}")
parts.append(
"(Earlier turn-by-turn details have been compacted to keep the "
"context window manageable. Continue from where you left off.)"
)
return "\n".join(parts)
def _trim_history_by_turns(messages: list[dict], max_messages: int) -> list[dict]:
"""Compact message history when it exceeds max_messages.
The Anthropic API requires every `tool_result` block to reference a
`tool_use_id` from a previous assistant message. Naive slicing can drop
a tool_use while keeping its tool_result, causing 400 errors. This
function avoids that by:
1. Walking forward to find a clean turn boundary (a fresh user-text
message that starts a new turn; no tool_result content).
2. Summarizing everything BEFORE that boundary into a single user-text
message and prepending it to the kept tail.
3. If no clean boundary exists at all, returning the original history
unchanged. Better to temporarily exceed the cap than to corrupt the
conversation and 400 every subsequent request.
The summary is built programmatically (no LLM call) from the message
structure: original task, tool call counts, recent ReportProgress brain
states, and last assistant text.
"""
if len(messages) <= max_messages:
return list(messages)
target_tail_size = max_messages - 1 # leave room for the summary message
cut_index: int | None = None
# First pass: walk forward looking for the EARLIEST clean cut point that
# gets us under the cap. This preserves the most recent detail.
for i in range(1, len(messages)):
if not _is_fresh_user_message(messages[i]):
continue
if len(messages) - i <= target_tail_size:
cut_index = i
break
# Second pass: if no cut point gets us under the cap (e.g. the current
# turn alone is bigger than max_messages), use the LATEST clean cut point
# available. The tail will still exceed the cap, but it's the smallest
# safe history we can produce; and any compaction is better than none.
if cut_index is None:
for i in range(len(messages) - 1, 0, -1):
if _is_fresh_user_message(messages[i]):
cut_index = i
break
if cut_index is None:
# No clean cut anywhere in the history. Return original; better to
# exceed the cap than to corrupt the conversation.
return list(messages)
# Compact: summarize messages[0..cut_index-1], prepend as a single
# user-text message, then keep messages[cut_index..end] verbatim.
summary_text = _summarize_messages(messages[:cut_index])
summary_msg = {"role": "user", "content": summary_text}
return [summary_msg] + list(messages[cut_index:])
BROWSER_TOOLS_SCHEMA = [
{
"name": "ReportProgress",
"description": (
"Record your assessment of the previous action and your plan for the "
"next one. You MUST call this BEFORE any browser action tools in every "
"turn (after the very first turn). This is how you reflect on what just "
"happened, track what you've learned about this site, and articulate what "
"you're trying to do next. Skipping it is not allowed and will be rejected."
),
"input_schema": {
"type": "object",
"properties": {
"evaluation_previous": {
"type": "string",
"description": (
"What did the previous action(s) accomplish? Did they succeed? "
"If not, why? Be specific about what changed on the page."
),
},
"working_memory": {
"type": "string",
"description": (
"Short notes about what you've learned about this site so far; "
"selectors that work, keyboard shortcuts, layout quirks, what "
"you've tried that failed. Carry this forward across turns."
),
},
"next_goal": {
"type": "string",
"description": (
"What you're trying to achieve with the action(s) you're about "
"to take next. Be concrete."
),
},
},
"required": ["evaluation_previous", "working_memory", "next_goal"],
},
},
{
"name": "BrowserScreenshot",
"description": (
"Capture a screenshot of the browser page. Returns the screenshot as a "
"base64-encoded PNG image. Use this to see what is currently displayed."
),
"input_schema": {
"type": "object",
"properties": {},
"required": [],
},
},
{
"name": "BrowserGetText",
"description": (
"Get the visible text content of the browser page. Returns up to 15000 characters."
),
"input_schema": {
"type": "object",
"properties": {},
"required": [],
},
},
{
"name": "BrowserNavigate",
"description": "Navigate the browser to a URL.",
"input_schema": {
"type": "object",
"properties": {
"url": {"type": "string", "description": "The URL to navigate to."},
},
"required": ["url"],
},
},
{
"name": "BrowserClick",
"description": "Click an element identified by a CSS selector. Use BrowserGetElements first to discover valid selectors.",
"input_schema": {
"type": "object",
"properties": {
"selector": {"type": "string", "description": "CSS selector of the element to click."},
},
"required": ["selector"],
},
},
{
"name": "BrowserType",
"description": "Type text into an input element. Clears existing value first.",
"input_schema": {
"type": "object",
"properties": {
"selector": {"type": "string", "description": "CSS selector of the input element."},
"text": {"type": "string", "description": "The text to type."},
},
"required": ["selector", "text"],
},
},
{
"name": "BrowserEvaluate",
"description": "Evaluate a JavaScript expression in the browser page and return the result.",
"input_schema": {
"type": "object",
"properties": {
"expression": {"type": "string", "description": "JavaScript expression to evaluate."},
},
"required": ["expression"],
},
},
{
"name": "BrowserGetElements",
"description": (
"Get a list of interactive elements on the page with CSS selectors. "
"Call this BEFORE clicking or typing so you know which selectors are valid."
),
"input_schema": {
"type": "object",
"properties": {
"selector": {
"type": "string",
"description": "Optional CSS selector to scope the search (e.g. 'form', '#main'). Defaults to 'body'.",
},
},
"required": [],
},
},
{
"name": "BrowserScroll",
"description": (
"Scroll the page up or down. Automatically finds the correct scrollable "
"container (works on SPAs like Notion, Gmail, etc. that use nested scroll "
"containers instead of window-level scrolling). Returns scroll position info "
"including whether top/bottom has been reached."
),
"input_schema": {
"type": "object",
"properties": {
"direction": {
"type": "string",
"enum": ["up", "down"],
"description": "Scroll direction. Defaults to 'down'.",
},
"amount": {
"type": "number",
"description": "Pixels to scroll. Defaults to 500.",
},
},
"required": [],
},
},
{
"name": "BrowserListInteractives",
"description": (
"Get a NUMBERED LIST of interactive elements on the page using the "
"browser's accessibility tree. Returns elements like [1]<button \"Like\">, "
"[2]<link \"Settings\">, etc. Use this BEFORE BrowserClickIndex. This is "
"the PREFERRED way to discover clickable elements on hostile sites "
"(Tinder, Instagram, TikTok) where CSS selectors fail because the page "
"uses unlabeled <div>s; the accessibility tree sees roles and names "
"even when raw HTML doesn't expose them. Much more reliable than "
"BrowserGetElements (which uses CSS selectors)."
),
"input_schema": {
"type": "object",
"properties": {},
"required": [],
},
},
{
"name": "BrowserClickIndex",
"description": (
"Click an element by its numeric index from BrowserListInteractives. "
"Uses native OS-level mouse events (event.isTrusted=true) so it works "
"on sites that filter out synthetic JS events. Always call "
"BrowserListInteractives first to get a fresh index list. If the click "
"returns 'index no longer valid', the page changed; re-list and retry."
),
"input_schema": {
"type": "object",
"properties": {
"index": {
"type": "integer",
"description": "The numeric index from BrowserListInteractives (1-based).",
},
},
"required": ["index"],
},
},
{
"name": "BrowserBatch",
"description": (
"Run a sequence of browser actions in one tool call. Each sub-action "
"is executed in order, with the URL captured before/after each one. "
"If the URL changes mid-batch (the page navigated), the rest of the "
"batch is aborted and you get a partial result. Use this when you "
"have a known sequence; typing then pressing Enter, swiping multiple "
"times, clicking through pagination. Max 5 actions per batch.\n\n"
"Sub-action types and their params:\n"
"- click_index: { index: int }\n"
"- press_key: { key: str }\n"
"- type: { selector: str, text: str }\n"
"- click: { selector: str }\n"
"- scroll: { direction?: 'up'|'down', amount?: int }\n"
"- wait: { milliseconds?: int }\n"
"- navigate: { url: str }\n\n"
"Example: { actions: [{type: 'click_index', params: {index: 1}}, "
"{type: 'wait', params: {milliseconds: 500}}, "
"{type: 'press_key', params: {key: 'ArrowRight'}}] }"
),
"input_schema": {
"type": "object",
"properties": {
"actions": {
"type": "array",
"maxItems": 5,
"items": {
"type": "object",
"properties": {
"type": {
"type": "string",
"enum": ["click_index", "press_key", "type", "wait", "scroll", "navigate", "click"],
},
"params": {"type": "object"},
},
"required": ["type", "params"],
},
},
},
"required": ["actions"],
},
},
{
"name": "BrowserPressKey",
"description": (
"Press a keyboard key (or key combination) on the page using a real native "
"input event. Use this for keyboard shortcuts when JS-dispatched events get "
"ignored; sites like Tinder, Slack, Notion, Gmail listen for trusted key "
"events. Examples: 'ArrowLeft', 'ArrowRight', 'Enter', 'Escape', 'Tab', "
"'Space', single letters like 'a'. Prefer this over BrowserEvaluate with "
"dispatchEvent for keyboard shortcuts."
),
"input_schema": {
"type": "object",
"properties": {
"key": {
"type": "string",
"description": (
"The key to press. Use JS KeyboardEvent.key names like "
"'ArrowUp', 'ArrowDown', 'Enter', 'Escape', 'Tab', 'Space', "
"'Backspace', or a single character like 'a'."
),
},
},
"required": ["key"],
},
},
{
"name": "BrowserWait",
"description": (
"Wait for a specified duration. Useful after navigation or actions that "
"trigger page loads, animations, or async content rendering. "
"Min 100ms, max 10000ms."
),
"input_schema": {
"type": "object",
"properties": {
"milliseconds": {
"type": "number",
"description": "Duration to wait in milliseconds. Defaults to 1000.",
},
},
"required": [],
},
},
{
"name": "RequestHumanIntervention",
"description": (
"Request the user's help when you encounter an obstacle you cannot solve "
"programmatically; captchas, login prompts, cookie consent walls, "
"two-factor authentication, or any blocking popup. The agent will pause "
"until the user resolves the issue and clicks Continue."
),
"input_schema": {
"type": "object",
"properties": {
"problem": {
"type": "string",
"description": (
"One short sentence describing the obstacle. Keep it under "
"15 words. Example: 'Login required; please sign in to X/Twitter.'"
),
},
"instruction": {
"type": "string",
"description": (
"One short sentence telling the user what to do. Keep it under "
"15 words. Example: 'Log in with your credentials, then click Done.'"
),
},
},
"required": ["problem", "instruction"],
},
},
]
ACTION_MAP = {
"BrowserScreenshot": "screenshot",
"BrowserGetText": "get_text",
"BrowserNavigate": "navigate",
"BrowserClick": "click",
"BrowserType": "type",
"BrowserEvaluate": "evaluate",
"BrowserGetElements": "get_elements",
"BrowserScroll": "scroll",
"BrowserWait": "wait",
"BrowserPressKey": "press_key",
"BrowserListInteractives": "list_interactives",
"BrowserClickIndex": "click_index",
"BrowserBatch": "batch",
}
SYSTEM_PROMPT = (
"You are a website-agnostic browser automation agent. You can operate on ANY "
"website the user is signed into; social media, dating apps, email, productivity "
"tools, dashboards, ecommerce, anything. Assume the user has already logged in.\n\n"
"## Required output structure: ReportProgress before every action\n"
"Before ANY action tool (BrowserClick, BrowserType, BrowserNavigate, "
"BrowserPressKey, BrowserScroll, BrowserEvaluate, BrowserClickIndex, "
"BrowserBatch), you MUST call the ReportProgress tool in the SAME turn. "
"ReportProgress takes three short fields:\n"
"- evaluation_previous: did your last action work? what changed on the page?\n"
"- working_memory: what have you learned about this site? what worked, what didn't?\n"
"- next_goal: what specifically are you trying to do with the next action?\n"
"Emit ReportProgress and your action tool(s) together in the same response. "
"If you skip ReportProgress, your action tools will be REJECTED with an error "
"and you will have to retry. This is not optional. Read-only tools "
"(BrowserScreenshot, BrowserGetText, BrowserGetElements, BrowserWait) do not "
"require ReportProgress.\n\n"
"## Loop awareness\n"
"If you see a tool result containing 'LOOP DETECTED' or '⚠️', it means you "
"have called the same tool with the same parameters and gotten the same "
"result multiple times in a row. STOP. Do NOT retry the same approach. "
"Switch strategy entirely: try a different tool, a different selector, "
"keyboard shortcuts, or call RequestHumanIntervention if you genuinely "
"cannot proceed. The loop detector will force-exit the agent if you "
"ignore it more than 5 times.\n\n"
"## Use prior context\n"
"If this is a continuation of an earlier conversation on the same browser, the "
"messages above already contain everything you've tried, what worked, what failed, "
"and the page state. READ THAT HISTORY before acting. Do NOT take a fresh screenshot "
"or re-explore the DOM if you already know what's on screen; just act. Only re-orient "
"if the page has clearly changed (after navigation, after a multi-second wait, or if "
"your last action mutated the page in unexpected ways).\n\n"
"## Try multiple strategies, learn from failures\n"
"Sites vary wildly. When one approach fails, switch tactics; don't retry the same "
"thing. The escalation ladder, fastest to slowest:\n"
"1. **Keyboard shortcuts via BrowserPressKey**; fastest and most reliable on sites "
"that support them (Tinder swipes, Gmail navigation, Slack message jump, etc.). "
"Always check if the site shows keyboard hints in the UI before falling back to clicks. "
"BrowserPressKey sends real native events that pass the `event.isTrusted` check, so "
"it works where dispatchEvent in BrowserEvaluate silently fails.\n"
"2. **Accessibility tree via BrowserListInteractives + BrowserClickIndex**; the "
"accessibility tree sees roles and names that the raw DOM doesn't, even on sites "
"like Tinder, Instagram, and TikTok that use unlabeled <div>s with click handlers. "
"Call BrowserListInteractives to get a numbered list (`[1]<button \"Like\">`, "
"`[2]<link \"Settings\">`), then BrowserClickIndex with the number. The click uses "
"native OS-level mouse events so it works where DOM .click() doesn't. THIS IS YOUR "
"GO-TO STRATEGY for unlabeled or hostile sites; try this BEFORE BrowserGetElements.\n"
"3. **Semantic CSS selectors**; `button[aria-label='X']`, `[role='button']`, "
"`a[href*='...']`. Try these via BrowserGetElements + BrowserClick when the site "
"actually has semantic HTML.\n"
"4. **Text-based JS query**; when both of the above fail, use BrowserEvaluate to "
"find elements by visible text: `Array.from(document.querySelectorAll('*')).find(el => el.textContent.trim() === 'Like')`.\n"
"5. **Coordinate-based fallback**; last resort: take a screenshot, identify the "
"button visually, then click by approximate coords.\n\n"
"## Batch known sequences with BrowserBatch\n"
"When you have a known sequence of actions; typing then pressing Enter, "
"swiping multiple times, clicking through pagination; emit them all in a "
"single BrowserBatch call instead of one tool per turn. The batch executes "
"sub-actions sequentially and aborts if the URL changes mid-batch (so you "
"won't operate on stale state). Max 5 sub-actions per batch.\n"
"Use BrowserBatch when:\n"
"- You're doing the same action repeatedly (5 swipes, 3 scrolls)\n"
"- You have a deterministic flow (type query → press Enter → click first result)\n"
"Don't use BrowserBatch when:\n"
"- You need to read the page state between actions\n"
"- You're uncertain about what comes next\n"
"- An action might trigger an unexpected popup or navigation\n\n"
"## Avoid wasted cycles\n"
"- Do NOT screenshot after every single action. Screenshot ONLY when you genuinely "
"don't know the page state (start of task, after navigation, after a failure).\n"
"- Do NOT call BrowserGetElements on the entire body if you already know roughly "
"where the target is. Scope it: `BrowserGetElements({selector: 'nav'})`.\n"
"- Do NOT call the same failing tool twice with identical parameters. If selector "
"X failed, try a DIFFERENT selector or a DIFFERENT strategy.\n"
"- For repeated actions (swiping through profiles, going through inbox messages), "
"use BrowserPressKey if available; it's an order of magnitude faster than DOM clicks.\n\n"
"## When you genuinely cannot proceed\n"
"Use RequestHumanIntervention for:\n"
"- Login walls (the user thinks they're logged in but the session expired)\n"
"- Captchas, 2FA prompts, age verification gates\n"
"- Anything genuinely ambiguous about user intent\n"
"Don't use it for normal tool failures; try a different approach first.\n\n"
"## Tool reference\n"
"- BrowserScreenshot: visual snapshot. Use sparingly, not after every action.\n"
"- BrowserGetText: returns up to 15000 chars of visible text. Useful for reading "
"content without an image.\n"
"- BrowserScroll: handles nested scroll containers (Notion, Gmail). Returns "
"atTop/atBottom; stop looping when scroll delta is 0.\n"
"- BrowserGetElements: enumerate interactive elements with selectors.\n"
"- BrowserClick / BrowserType: standard DOM interaction.\n"
"- BrowserPressKey: native key events (preferred for shortcuts).\n"
"- BrowserEvaluate: arbitrary JS for everything else, including text-based element "
"search and reading state. Avoid for scrolling and keyboard events.\n"
"- BrowserWait: 1-3s after navigation, 0.5s after most clicks.\n\n"
"Complete the task autonomously and report a clear, brief summary."
)
MAX_TURNS = 40
# Tools that count as "action tools"; calling any of these in a turn requires
# the model to also call ReportProgress in the same turn (after the first
# turn). Read-only tools and meta tools are exempt.
_ACTION_TOOLS_REQUIRING_REPORT = {
"BrowserClick",
"BrowserType",
"BrowserNavigate",
"BrowserPressKey",
"BrowserScroll",
"BrowserEvaluate",
"BrowserClickIndex", # Phase 3
"BrowserBatch", # Phase 4
}
async def execute_browser_tool(
tool_name: str, tool_input: dict, browser_id: str, tab_id: str = "",
@@ -955,13 +255,13 @@ async def run_browser_agent(
# cycle every time the parent issues a new task. Defensively validate
# the cache; if it's somehow corrupted (orphaned tool_use_ids), drop
# it and start fresh rather than crash on the next API call.
prior_messages = _browser_history.get(browser_id) or []
prior_messages = browser_history._browser_history.get(browser_id) or []
if prior_messages and not _validate_message_pairing(prior_messages):
logger.warning(
f"[browser-agent {session_id}] cached history for {browser_id} has "
f"orphaned tool_use_ids; dropping cache and starting fresh"
)
_browser_history.pop(browser_id, None)
clear_browser_history(browser_id)
prior_messages = []
messages: list[dict] = list(prior_messages) + [{"role": "user", "content": task}]
action_log: list[dict] = []
@@ -1379,7 +679,7 @@ async def run_browser_agent(
# _MAX_HISTORY_MESSAGES turns to keep token usage bounded; but
# never split a tool_use ↔ tool_result pair across the cut, or the
# next API request will 400.
_browser_history[browser_id] = _trim_history_by_turns(
browser_history._browser_history[browser_id] = _trim_history_by_turns(
messages, _MAX_HISTORY_MESSAGES,
)
+209
View File
@@ -0,0 +1,209 @@
"""
Conversation-history cache for the browser sub-agent.
Caches conversation history per browser_id so successive BrowserAgent calls on
the same browser can resume rather than restart from scratch. Without this every
"swipe right" / "swipe left" call has to take a new screenshot and re-orient
itself, costing 30-60s per action.
The `_browser_history` mutable cache lives in EXACTLY this module; all reads and
writes route through here so there's a single source of truth.
"""
# browser_id -> cached Anthropic message list for resume.
_browser_history: dict[str, list[dict]] = {}
# Cap history to prevent unbounded growth on long-lived browsers.
_MAX_HISTORY_MESSAGES = 30
def clear_browser_history(browser_id: str) -> None:
"""Drop cached conversation history for a browser (e.g. when it's closed)."""
_browser_history.pop(browser_id, None)
def _validate_message_pairing(messages: list[dict]) -> bool:
"""Verify every tool_result references a tool_use_id from a prior assistant
message in the same list. Returns False if there's an orphan, which means
the cached history would 400 if sent to the API.
This is the last line of defense against cache corruption; if it ever
returns False on a resume, we drop the cache and start fresh rather than
crash on the next API call.
"""
declared_tool_use_ids: set[str] = set()
for msg in messages:
role = msg.get("role")
content = msg.get("content")
if role == "assistant" and isinstance(content, list):
for block in content:
if isinstance(block, dict) and block.get("type") == "tool_use":
tu_id = block.get("id")
if tu_id:
declared_tool_use_ids.add(tu_id)
elif role == "user" and isinstance(content, list):
for block in content:
if isinstance(block, dict) and block.get("type") == "tool_result":
tr_id = block.get("tool_use_id")
if tr_id and tr_id not in declared_tool_use_ids:
return False
return True
def _is_fresh_user_message(msg: dict) -> bool:
"""A 'fresh' user message starts a new turn; string content or a list
that contains no tool_result blocks. These are the only safe cut points
because they don't reference any prior assistant tool_use blocks."""
if msg.get("role") != "user":
return False
content = msg.get("content")
if isinstance(content, str):
return True
if isinstance(content, list) and not any(
isinstance(c, dict) and c.get("type") == "tool_result" for c in content
):
return True
return False
def _summarize_messages(messages: list[dict]) -> str:
"""Build a programmatic summary of older browser-agent messages.
Extracts the original user task, a count of tool calls by name with their
key parameters, the last few ReportProgress brain states, and the most
recent assistant text. No LLM call required; this is purely structural
extraction from the existing message history.
"""
if not messages:
return ""
# Find the original user task (first user-text message)
initial_task = ""
for msg in messages:
if msg.get("role") == "user":
content = msg.get("content")
if isinstance(content, str) and content.strip():
initial_task = content.strip()[:300]
break
# Count tool calls by name with key params
tool_call_summary: dict[str, list[str]] = {}
brain_states: list[str] = []
last_assistant_text = ""
for msg in messages:
if msg.get("role") != "assistant":
continue
content = msg.get("content")
if not isinstance(content, list):
continue
for block in content:
if not isinstance(block, dict):
continue
btype = block.get("type")
if btype == "tool_use":
name = block.get("name", "unknown")
inp = block.get("input") or {}
if name == "ReportProgress":
# Capture the brain state for inline summary
brain_states.append(
f"{inp.get('next_goal', '')[:120]}"
)
continue
# Compact one-line description with key params
key_param = ""
for k in ("index", "key", "url", "selector", "direction", "text"):
if k in inp:
v = str(inp[k])[:40]
key_param = f"{k}={v}"
break
desc = f"{name}({key_param})" if key_param else name
tool_call_summary.setdefault(name, []).append(desc)
elif btype == "text":
txt = block.get("text", "").strip()
if txt:
last_assistant_text = txt
# Build the summary text
parts = ["[Summary of earlier browser-agent activity]"]
if initial_task:
parts.append(f'Original task: "{initial_task}"')
if tool_call_summary:
total = sum(len(v) for v in tool_call_summary.values())
parts.append(f"Actions taken ({total} total):")
# Show count + a couple of representative examples per tool
for name in sorted(tool_call_summary.keys()):
calls = tool_call_summary[name]
count = len(calls)
sample = calls[-1] # most recent example
if count == 1:
parts.append(f" - {sample}")
else:
parts.append(f" - {sample} (×{count})")
if brain_states:
parts.append("Recent intents:")
parts.extend(brain_states[-5:]) # last 5 brain states
if last_assistant_text:
snippet = last_assistant_text[:400]
parts.append(f"Last update from assistant: {snippet}")
parts.append(
"(Earlier turn-by-turn details have been compacted to keep the "
"context window manageable. Continue from where you left off.)"
)
return "\n".join(parts)
def _trim_history_by_turns(messages: list[dict], max_messages: int) -> list[dict]:
"""Compact message history when it exceeds max_messages.
The Anthropic API requires every `tool_result` block to reference a
`tool_use_id` from a previous assistant message. Naive slicing can drop
a tool_use while keeping its tool_result, causing 400 errors. This
function avoids that by:
1. Walking forward to find a clean turn boundary (a fresh user-text
message that starts a new turn; no tool_result content).
2. Summarizing everything BEFORE that boundary into a single user-text
message and prepending it to the kept tail.
3. If no clean boundary exists at all, returning the original history
unchanged. Better to temporarily exceed the cap than to corrupt the
conversation and 400 every subsequent request.
The summary is built programmatically (no LLM call) from the message
structure: original task, tool call counts, recent ReportProgress brain
states, and last assistant text.
"""
if len(messages) <= max_messages:
return list(messages)
target_tail_size = max_messages - 1 # leave room for the summary message
cut_index: int | None = None
# First pass: walk forward looking for the EARLIEST clean cut point that
# gets us under the cap. This preserves the most recent detail.
for i in range(1, len(messages)):
if not _is_fresh_user_message(messages[i]):
continue
if len(messages) - i <= target_tail_size:
cut_index = i
break
# Second pass: if no cut point gets us under the cap (e.g. the current
# turn alone is bigger than max_messages), use the LATEST clean cut point
# available. The tail will still exceed the cap, but it's the smallest
# safe history we can produce; and any compaction is better than none.
if cut_index is None:
for i in range(len(messages) - 1, 0, -1):
if _is_fresh_user_message(messages[i]):
cut_index = i
break
if cut_index is None:
# No clean cut anywhere in the history. Return original; better to
# exceed the cap than to corrupt the conversation.
return list(messages)
# Compact: summarize messages[0..cut_index-1], prepend as a single
# user-text message, then keep messages[cut_index..end] verbatim.
summary_text = _summarize_messages(messages[:cut_index])
summary_msg = {"role": "user", "content": summary_text}
return [summary_msg] + list(messages[cut_index:])
+74
View File
@@ -0,0 +1,74 @@
"""
Loop detection for the browser sub-agent.
Tracks recent state-mutating tool calls in a sliding window. If the model
repeats the same (tool, input) with the same result several times, we inject
an is_error message in the next tool_result to force a strategy change. This
prevents the model from burning the entire turn budget on a failing approach.
"""
import json
# Tools that are read-only / idempotent and should NOT count toward loop
# detection. Repeating these is normal (scrolling through a feed, taking
# successive screenshots, polling for an element to appear).
_LOOP_DETECTION_EXCLUDED_TOOLS = {
"BrowserScreenshot",
"BrowserGetText",
"BrowserGetElements",
"BrowserListInteractives", # Phase 3
"BrowserWait",
"ReportProgress", # Phase 2
"RequestHumanIntervention",
}
_LOOP_WINDOW_SIZE = 5
_LOOP_REPEAT_THRESHOLD = 3
_LOOP_HARD_CAP = 5
def _hash_tool_call(tool_name: str, tool_input: dict, result: dict) -> tuple[str, str, str]:
"""Build a stable hash key for a tool call, including its result.
Including the result hash means that legitimate progress (same input,
different output; e.g. BrowserScroll on a long feed) does NOT count
as a loop. Only same-input + same-output is treated as stuck.
"""
try:
input_key = json.dumps(tool_input, sort_keys=True, default=str)
except Exception:
input_key = repr(tool_input)
try:
# Truncate the result hash to avoid huge image blobs in the key
result_key = json.dumps(result, sort_keys=True, default=str)[:300]
except Exception:
result_key = repr(result)[:300]
return (tool_name, input_key, result_key)
def _detect_loop(
recent_calls: list[tuple[str, str, str]],
new_call: tuple[str, str, str],
) -> bool:
"""Return True if `new_call` constitutes a loop given recent history.
A loop is when the same (tool, input, result) has appeared at least
`_LOOP_REPEAT_THRESHOLD` times within the last `_LOOP_WINDOW_SIZE`
state-mutating calls (the new call counts as one of those occurrences).
"""
if new_call[0] in _LOOP_DETECTION_EXCLUDED_TOOLS:
return False
window = recent_calls[-(_LOOP_WINDOW_SIZE - 1):] + [new_call]
matches = sum(1 for c in window if c == new_call)
return matches >= _LOOP_REPEAT_THRESHOLD
_LOOP_WARNING_TEXT = (
"LOOP DETECTED: You have called this tool with these exact parameters and "
"gotten the same result {count} times in a row. STOP retrying this approach "
", it is not working. Try a fundamentally different strategy: "
"(1) check the page state with BrowserScreenshot or BrowserGetText, "
"(2) try a different selector or a different tool, "
"(3) use BrowserPressKey for keyboard shortcuts if the site supports them, "
"or (4) call RequestHumanIntervention if you genuinely cannot proceed."
)
+454
View File
@@ -0,0 +1,454 @@
"""
Static schema + prompt blob for the browser sub-agent.
One big single-responsibility constant file: tool schema, action map, system
prompt, and the turn/report invariants. Exceeds the 300-LOC soft ceiling on
purpose because it is one cohesive data blob, not multiple responsibilities.
"""
MODEL_MAP = {
"sonnet": "claude-sonnet-4-6",
"opus": "claude-opus-4-6",
"haiku": "claude-haiku-4-5-20251001",
}
BROWSER_TOOLS_SCHEMA = [
{
"name": "ReportProgress",
"description": (
"Record your assessment of the previous action and your plan for the "
"next one. You MUST call this BEFORE any browser action tools in every "
"turn (after the very first turn). This is how you reflect on what just "
"happened, track what you've learned about this site, and articulate what "
"you're trying to do next. Skipping it is not allowed and will be rejected."
),
"input_schema": {
"type": "object",
"properties": {
"evaluation_previous": {
"type": "string",
"description": (
"What did the previous action(s) accomplish? Did they succeed? "
"If not, why? Be specific about what changed on the page."
),
},
"working_memory": {
"type": "string",
"description": (
"Short notes about what you've learned about this site so far; "
"selectors that work, keyboard shortcuts, layout quirks, what "
"you've tried that failed. Carry this forward across turns."
),
},
"next_goal": {
"type": "string",
"description": (
"What you're trying to achieve with the action(s) you're about "
"to take next. Be concrete."
),
},
},
"required": ["evaluation_previous", "working_memory", "next_goal"],
},
},
{
"name": "BrowserScreenshot",
"description": (
"Capture a screenshot of the browser page. Returns the screenshot as a "
"base64-encoded PNG image. Use this to see what is currently displayed."
),
"input_schema": {
"type": "object",
"properties": {},
"required": [],
},
},
{
"name": "BrowserGetText",
"description": (
"Get the visible text content of the browser page. Returns up to 15000 characters."
),
"input_schema": {
"type": "object",
"properties": {},
"required": [],
},
},
{
"name": "BrowserNavigate",
"description": "Navigate the browser to a URL.",
"input_schema": {
"type": "object",
"properties": {
"url": {"type": "string", "description": "The URL to navigate to."},
},
"required": ["url"],
},
},
{
"name": "BrowserClick",
"description": "Click an element identified by a CSS selector. Use BrowserGetElements first to discover valid selectors.",
"input_schema": {
"type": "object",
"properties": {
"selector": {"type": "string", "description": "CSS selector of the element to click."},
},
"required": ["selector"],
},
},
{
"name": "BrowserType",
"description": "Type text into an input element. Clears existing value first.",
"input_schema": {
"type": "object",
"properties": {
"selector": {"type": "string", "description": "CSS selector of the input element."},
"text": {"type": "string", "description": "The text to type."},
},
"required": ["selector", "text"],
},
},
{
"name": "BrowserEvaluate",
"description": "Evaluate a JavaScript expression in the browser page and return the result.",
"input_schema": {
"type": "object",
"properties": {
"expression": {"type": "string", "description": "JavaScript expression to evaluate."},
},
"required": ["expression"],
},
},
{
"name": "BrowserGetElements",
"description": (
"Get a list of interactive elements on the page with CSS selectors. "
"Call this BEFORE clicking or typing so you know which selectors are valid."
),
"input_schema": {
"type": "object",
"properties": {
"selector": {
"type": "string",
"description": "Optional CSS selector to scope the search (e.g. 'form', '#main'). Defaults to 'body'.",
},
},
"required": [],
},
},
{
"name": "BrowserScroll",
"description": (
"Scroll the page up or down. Automatically finds the correct scrollable "
"container (works on SPAs like Notion, Gmail, etc. that use nested scroll "
"containers instead of window-level scrolling). Returns scroll position info "
"including whether top/bottom has been reached."
),
"input_schema": {
"type": "object",
"properties": {
"direction": {
"type": "string",
"enum": ["up", "down"],
"description": "Scroll direction. Defaults to 'down'.",
},
"amount": {
"type": "number",
"description": "Pixels to scroll. Defaults to 500.",
},
},
"required": [],
},
},
{
"name": "BrowserListInteractives",
"description": (
"Get a NUMBERED LIST of interactive elements on the page using the "
"browser's accessibility tree. Returns elements like [1]<button \"Like\">, "
"[2]<link \"Settings\">, etc. Use this BEFORE BrowserClickIndex. This is "
"the PREFERRED way to discover clickable elements on hostile sites "
"(Tinder, Instagram, TikTok) where CSS selectors fail because the page "
"uses unlabeled <div>s; the accessibility tree sees roles and names "
"even when raw HTML doesn't expose them. Much more reliable than "
"BrowserGetElements (which uses CSS selectors)."
),
"input_schema": {
"type": "object",
"properties": {},
"required": [],
},
},
{
"name": "BrowserClickIndex",
"description": (
"Click an element by its numeric index from BrowserListInteractives. "
"Uses native OS-level mouse events (event.isTrusted=true) so it works "
"on sites that filter out synthetic JS events. Always call "
"BrowserListInteractives first to get a fresh index list. If the click "
"returns 'index no longer valid', the page changed; re-list and retry."
),
"input_schema": {
"type": "object",
"properties": {
"index": {
"type": "integer",
"description": "The numeric index from BrowserListInteractives (1-based).",
},
},
"required": ["index"],
},
},
{
"name": "BrowserBatch",
"description": (
"Run a sequence of browser actions in one tool call. Each sub-action "
"is executed in order, with the URL captured before/after each one. "
"If the URL changes mid-batch (the page navigated), the rest of the "
"batch is aborted and you get a partial result. Use this when you "
"have a known sequence; typing then pressing Enter, swiping multiple "
"times, clicking through pagination. Max 5 actions per batch.\n\n"
"Sub-action types and their params:\n"
"- click_index: { index: int }\n"
"- press_key: { key: str }\n"
"- type: { selector: str, text: str }\n"
"- click: { selector: str }\n"
"- scroll: { direction?: 'up'|'down', amount?: int }\n"
"- wait: { milliseconds?: int }\n"
"- navigate: { url: str }\n\n"
"Example: { actions: [{type: 'click_index', params: {index: 1}}, "
"{type: 'wait', params: {milliseconds: 500}}, "
"{type: 'press_key', params: {key: 'ArrowRight'}}] }"
),
"input_schema": {
"type": "object",
"properties": {
"actions": {
"type": "array",
"maxItems": 5,
"items": {
"type": "object",
"properties": {
"type": {
"type": "string",
"enum": ["click_index", "press_key", "type", "wait", "scroll", "navigate", "click"],
},
"params": {"type": "object"},
},
"required": ["type", "params"],
},
},
},
"required": ["actions"],
},
},
{
"name": "BrowserPressKey",
"description": (
"Press a keyboard key (or key combination) on the page using a real native "
"input event. Use this for keyboard shortcuts when JS-dispatched events get "
"ignored; sites like Tinder, Slack, Notion, Gmail listen for trusted key "
"events. Examples: 'ArrowLeft', 'ArrowRight', 'Enter', 'Escape', 'Tab', "
"'Space', single letters like 'a'. Prefer this over BrowserEvaluate with "
"dispatchEvent for keyboard shortcuts."
),
"input_schema": {
"type": "object",
"properties": {
"key": {
"type": "string",
"description": (
"The key to press. Use JS KeyboardEvent.key names like "
"'ArrowUp', 'ArrowDown', 'Enter', 'Escape', 'Tab', 'Space', "
"'Backspace', or a single character like 'a'."
),
},
},
"required": ["key"],
},
},
{
"name": "BrowserWait",
"description": (
"Wait for a specified duration. Useful after navigation or actions that "
"trigger page loads, animations, or async content rendering. "
"Min 100ms, max 10000ms."
),
"input_schema": {
"type": "object",
"properties": {
"milliseconds": {
"type": "number",
"description": "Duration to wait in milliseconds. Defaults to 1000.",
},
},
"required": [],
},
},
{
"name": "RequestHumanIntervention",
"description": (
"Request the user's help when you encounter an obstacle you cannot solve "
"programmatically; captchas, login prompts, cookie consent walls, "
"two-factor authentication, or any blocking popup. The agent will pause "
"until the user resolves the issue and clicks Continue."
),
"input_schema": {
"type": "object",
"properties": {
"problem": {
"type": "string",
"description": (
"One short sentence describing the obstacle. Keep it under "
"15 words. Example: 'Login required; please sign in to X/Twitter.'"
),
},
"instruction": {
"type": "string",
"description": (
"One short sentence telling the user what to do. Keep it under "
"15 words. Example: 'Log in with your credentials, then click Done.'"
),
},
},
"required": ["problem", "instruction"],
},
},
]
ACTION_MAP = {
"BrowserScreenshot": "screenshot",
"BrowserGetText": "get_text",
"BrowserNavigate": "navigate",
"BrowserClick": "click",
"BrowserType": "type",
"BrowserEvaluate": "evaluate",
"BrowserGetElements": "get_elements",
"BrowserScroll": "scroll",
"BrowserWait": "wait",
"BrowserPressKey": "press_key",
"BrowserListInteractives": "list_interactives",
"BrowserClickIndex": "click_index",
"BrowserBatch": "batch",
}
SYSTEM_PROMPT = (
"You are a website-agnostic browser automation agent. You can operate on ANY "
"website the user is signed into; social media, dating apps, email, productivity "
"tools, dashboards, ecommerce, anything. Assume the user has already logged in.\n\n"
"## Required output structure: ReportProgress before every action\n"
"Before ANY action tool (BrowserClick, BrowserType, BrowserNavigate, "
"BrowserPressKey, BrowserScroll, BrowserEvaluate, BrowserClickIndex, "
"BrowserBatch), you MUST call the ReportProgress tool in the SAME turn. "
"ReportProgress takes three short fields:\n"
"- evaluation_previous: did your last action work? what changed on the page?\n"
"- working_memory: what have you learned about this site? what worked, what didn't?\n"
"- next_goal: what specifically are you trying to do with the next action?\n"
"Emit ReportProgress and your action tool(s) together in the same response. "
"If you skip ReportProgress, your action tools will be REJECTED with an error "
"and you will have to retry. This is not optional. Read-only tools "
"(BrowserScreenshot, BrowserGetText, BrowserGetElements, BrowserWait) do not "
"require ReportProgress.\n\n"
"## Loop awareness\n"
"If you see a tool result containing 'LOOP DETECTED' or '⚠️', it means you "
"have called the same tool with the same parameters and gotten the same "
"result multiple times in a row. STOP. Do NOT retry the same approach. "
"Switch strategy entirely: try a different tool, a different selector, "
"keyboard shortcuts, or call RequestHumanIntervention if you genuinely "
"cannot proceed. The loop detector will force-exit the agent if you "
"ignore it more than 5 times.\n\n"
"## Use prior context\n"
"If this is a continuation of an earlier conversation on the same browser, the "
"messages above already contain everything you've tried, what worked, what failed, "
"and the page state. READ THAT HISTORY before acting. Do NOT take a fresh screenshot "
"or re-explore the DOM if you already know what's on screen; just act. Only re-orient "
"if the page has clearly changed (after navigation, after a multi-second wait, or if "
"your last action mutated the page in unexpected ways).\n\n"
"## Try multiple strategies, learn from failures\n"
"Sites vary wildly. When one approach fails, switch tactics; don't retry the same "
"thing. The escalation ladder, fastest to slowest:\n"
"1. **Keyboard shortcuts via BrowserPressKey**; fastest and most reliable on sites "
"that support them (Tinder swipes, Gmail navigation, Slack message jump, etc.). "
"Always check if the site shows keyboard hints in the UI before falling back to clicks. "
"BrowserPressKey sends real native events that pass the `event.isTrusted` check, so "
"it works where dispatchEvent in BrowserEvaluate silently fails.\n"
"2. **Accessibility tree via BrowserListInteractives + BrowserClickIndex**; the "
"accessibility tree sees roles and names that the raw DOM doesn't, even on sites "
"like Tinder, Instagram, and TikTok that use unlabeled <div>s with click handlers. "
"Call BrowserListInteractives to get a numbered list (`[1]<button \"Like\">`, "
"`[2]<link \"Settings\">`), then BrowserClickIndex with the number. The click uses "
"native OS-level mouse events so it works where DOM .click() doesn't. THIS IS YOUR "
"GO-TO STRATEGY for unlabeled or hostile sites; try this BEFORE BrowserGetElements.\n"
"3. **Semantic CSS selectors**; `button[aria-label='X']`, `[role='button']`, "
"`a[href*='...']`. Try these via BrowserGetElements + BrowserClick when the site "
"actually has semantic HTML.\n"
"4. **Text-based JS query**; when both of the above fail, use BrowserEvaluate to "
"find elements by visible text: `Array.from(document.querySelectorAll('*')).find(el => el.textContent.trim() === 'Like')`.\n"
"5. **Coordinate-based fallback**; last resort: take a screenshot, identify the "
"button visually, then click by approximate coords.\n\n"
"## Batch known sequences with BrowserBatch\n"
"When you have a known sequence of actions; typing then pressing Enter, "
"swiping multiple times, clicking through pagination; emit them all in a "
"single BrowserBatch call instead of one tool per turn. The batch executes "
"sub-actions sequentially and aborts if the URL changes mid-batch (so you "
"won't operate on stale state). Max 5 sub-actions per batch.\n"
"Use BrowserBatch when:\n"
"- You're doing the same action repeatedly (5 swipes, 3 scrolls)\n"
"- You have a deterministic flow (type query → press Enter → click first result)\n"
"Don't use BrowserBatch when:\n"
"- You need to read the page state between actions\n"
"- You're uncertain about what comes next\n"
"- An action might trigger an unexpected popup or navigation\n\n"
"## Avoid wasted cycles\n"
"- Do NOT screenshot after every single action. Screenshot ONLY when you genuinely "
"don't know the page state (start of task, after navigation, after a failure).\n"
"- Do NOT call BrowserGetElements on the entire body if you already know roughly "
"where the target is. Scope it: `BrowserGetElements({selector: 'nav'})`.\n"
"- Do NOT call the same failing tool twice with identical parameters. If selector "
"X failed, try a DIFFERENT selector or a DIFFERENT strategy.\n"
"- For repeated actions (swiping through profiles, going through inbox messages), "
"use BrowserPressKey if available; it's an order of magnitude faster than DOM clicks.\n\n"
"## When you genuinely cannot proceed\n"
"Use RequestHumanIntervention for:\n"
"- Login walls (the user thinks they're logged in but the session expired)\n"
"- Captchas, 2FA prompts, age verification gates\n"
"- Anything genuinely ambiguous about user intent\n"
"Don't use it for normal tool failures; try a different approach first.\n\n"
"## Tool reference\n"
"- BrowserScreenshot: visual snapshot. Use sparingly, not after every action.\n"
"- BrowserGetText: returns up to 15000 chars of visible text. Useful for reading "
"content without an image.\n"
"- BrowserScroll: handles nested scroll containers (Notion, Gmail). Returns "
"atTop/atBottom; stop looping when scroll delta is 0.\n"
"- BrowserGetElements: enumerate interactive elements with selectors.\n"
"- BrowserClick / BrowserType: standard DOM interaction.\n"
"- BrowserPressKey: native key events (preferred for shortcuts).\n"
"- BrowserEvaluate: arbitrary JS for everything else, including text-based element "
"search and reading state. Avoid for scrolling and keyboard events.\n"
"- BrowserWait: 1-3s after navigation, 0.5s after most clicks.\n\n"
"Complete the task autonomously and report a clear, brief summary."
)
MAX_TURNS = 40
# Tools that count as "action tools"; calling any of these in a turn requires
# the model to also call ReportProgress in the same turn (after the first
# turn). Read-only tools and meta tools are exempt.
_ACTION_TOOLS_REQUIRING_REPORT = {
"BrowserClick",
"BrowserType",
"BrowserNavigate",
"BrowserPressKey",
"BrowserScroll",
"BrowserEvaluate",
"BrowserClickIndex", # Phase 3
"BrowserBatch", # Phase 4
}