mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-08-31 20:29:56 +02:00
[eric] split: carve browser_agent schema/history/loop into modules
This commit is contained in:
@@ -15,735 +15,35 @@ from uuid import uuid4
|
||||
|
||||
import anthropic
|
||||
|
||||
from backend.apps.agents import browser_history
|
||||
from backend.apps.agents.browser_history import (
|
||||
_MAX_HISTORY_MESSAGES,
|
||||
_trim_history_by_turns,
|
||||
_validate_message_pairing,
|
||||
clear_browser_history,
|
||||
)
|
||||
from backend.apps.agents.browser_loop import (
|
||||
_LOOP_DETECTION_EXCLUDED_TOOLS,
|
||||
_LOOP_HARD_CAP,
|
||||
_LOOP_WARNING_TEXT,
|
||||
_LOOP_WINDOW_SIZE,
|
||||
_detect_loop,
|
||||
_hash_tool_call,
|
||||
)
|
||||
from backend.apps.agents.browser_schema import (
|
||||
_ACTION_TOOLS_REQUIRING_REPORT,
|
||||
ACTION_MAP,
|
||||
BROWSER_TOOLS_SCHEMA,
|
||||
MAX_TURNS,
|
||||
MODEL_MAP,
|
||||
SYSTEM_PROMPT,
|
||||
)
|
||||
from backend.apps.agents.models import AgentSession, ApprovalRequest, Message
|
||||
from backend.apps.agents.ws_manager import ws_manager
|
||||
from backend.apps.tools_lib.tools_lib import load_builtin_permissions
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MODEL_MAP = {
|
||||
"sonnet": "claude-sonnet-4-6",
|
||||
"opus": "claude-opus-4-6",
|
||||
"haiku": "claude-haiku-4-5-20251001",
|
||||
}
|
||||
|
||||
# Cache of conversation history per browser_id so successive BrowserAgent
|
||||
# calls on the same browser can resume rather than restart from scratch.
|
||||
# Without this every "swipe right" / "swipe left" call has to take a new
|
||||
# screenshot and re-orient itself, costing 30-60s per action.
|
||||
_browser_history: dict[str, list[dict]] = {}
|
||||
# Cap history to prevent unbounded growth on long-lived browsers.
|
||||
_MAX_HISTORY_MESSAGES = 30
|
||||
|
||||
|
||||
def clear_browser_history(browser_id: str) -> None:
|
||||
"""Drop cached conversation history for a browser (e.g. when it's closed)."""
|
||||
_browser_history.pop(browser_id, None)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Loop detection
|
||||
#
|
||||
# Tracks recent state-mutating tool calls in a sliding window. If the model
|
||||
# repeats the same (tool, input) with the same result several times, we
|
||||
# inject an is_error message in the next tool_result to force a strategy
|
||||
# change. This prevents the model from burning the entire turn budget on
|
||||
# a failing approach.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Tools that are read-only / idempotent and should NOT count toward loop
|
||||
# detection. Repeating these is normal (scrolling through a feed, taking
|
||||
# successive screenshots, polling for an element to appear).
|
||||
_LOOP_DETECTION_EXCLUDED_TOOLS = {
|
||||
"BrowserScreenshot",
|
||||
"BrowserGetText",
|
||||
"BrowserGetElements",
|
||||
"BrowserListInteractives", # Phase 3
|
||||
"BrowserWait",
|
||||
"ReportProgress", # Phase 2
|
||||
"RequestHumanIntervention",
|
||||
}
|
||||
|
||||
_LOOP_WINDOW_SIZE = 5
|
||||
_LOOP_REPEAT_THRESHOLD = 3
|
||||
_LOOP_HARD_CAP = 5
|
||||
|
||||
|
||||
def _hash_tool_call(tool_name: str, tool_input: dict, result: dict) -> tuple[str, str, str]:
|
||||
"""Build a stable hash key for a tool call, including its result.
|
||||
|
||||
Including the result hash means that legitimate progress (same input,
|
||||
different output; e.g. BrowserScroll on a long feed) does NOT count
|
||||
as a loop. Only same-input + same-output is treated as stuck.
|
||||
"""
|
||||
try:
|
||||
input_key = json.dumps(tool_input, sort_keys=True, default=str)
|
||||
except Exception:
|
||||
input_key = repr(tool_input)
|
||||
try:
|
||||
# Truncate the result hash to avoid huge image blobs in the key
|
||||
result_key = json.dumps(result, sort_keys=True, default=str)[:300]
|
||||
except Exception:
|
||||
result_key = repr(result)[:300]
|
||||
return (tool_name, input_key, result_key)
|
||||
|
||||
|
||||
def _detect_loop(
|
||||
recent_calls: list[tuple[str, str, str]],
|
||||
new_call: tuple[str, str, str],
|
||||
) -> bool:
|
||||
"""Return True if `new_call` constitutes a loop given recent history.
|
||||
|
||||
A loop is when the same (tool, input, result) has appeared at least
|
||||
`_LOOP_REPEAT_THRESHOLD` times within the last `_LOOP_WINDOW_SIZE`
|
||||
state-mutating calls (the new call counts as one of those occurrences).
|
||||
"""
|
||||
if new_call[0] in _LOOP_DETECTION_EXCLUDED_TOOLS:
|
||||
return False
|
||||
window = recent_calls[-(_LOOP_WINDOW_SIZE - 1):] + [new_call]
|
||||
matches = sum(1 for c in window if c == new_call)
|
||||
return matches >= _LOOP_REPEAT_THRESHOLD
|
||||
|
||||
|
||||
_LOOP_WARNING_TEXT = (
|
||||
"LOOP DETECTED: You have called this tool with these exact parameters and "
|
||||
"gotten the same result {count} times in a row. STOP retrying this approach "
|
||||
", it is not working. Try a fundamentally different strategy: "
|
||||
"(1) check the page state with BrowserScreenshot or BrowserGetText, "
|
||||
"(2) try a different selector or a different tool, "
|
||||
"(3) use BrowserPressKey for keyboard shortcuts if the site supports them, "
|
||||
"or (4) call RequestHumanIntervention if you genuinely cannot proceed."
|
||||
)
|
||||
|
||||
|
||||
def _validate_message_pairing(messages: list[dict]) -> bool:
|
||||
"""Verify every tool_result references a tool_use_id from a prior assistant
|
||||
message in the same list. Returns False if there's an orphan, which means
|
||||
the cached history would 400 if sent to the API.
|
||||
|
||||
This is the last line of defense against cache corruption; if it ever
|
||||
returns False on a resume, we drop the cache and start fresh rather than
|
||||
crash on the next API call.
|
||||
"""
|
||||
declared_tool_use_ids: set[str] = set()
|
||||
for msg in messages:
|
||||
role = msg.get("role")
|
||||
content = msg.get("content")
|
||||
if role == "assistant" and isinstance(content, list):
|
||||
for block in content:
|
||||
if isinstance(block, dict) and block.get("type") == "tool_use":
|
||||
tu_id = block.get("id")
|
||||
if tu_id:
|
||||
declared_tool_use_ids.add(tu_id)
|
||||
elif role == "user" and isinstance(content, list):
|
||||
for block in content:
|
||||
if isinstance(block, dict) and block.get("type") == "tool_result":
|
||||
tr_id = block.get("tool_use_id")
|
||||
if tr_id and tr_id not in declared_tool_use_ids:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _is_fresh_user_message(msg: dict) -> bool:
|
||||
"""A 'fresh' user message starts a new turn; string content or a list
|
||||
that contains no tool_result blocks. These are the only safe cut points
|
||||
because they don't reference any prior assistant tool_use blocks."""
|
||||
if msg.get("role") != "user":
|
||||
return False
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str):
|
||||
return True
|
||||
if isinstance(content, list) and not any(
|
||||
isinstance(c, dict) and c.get("type") == "tool_result" for c in content
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _summarize_messages(messages: list[dict]) -> str:
|
||||
"""Build a programmatic summary of older browser-agent messages.
|
||||
|
||||
Extracts the original user task, a count of tool calls by name with their
|
||||
key parameters, the last few ReportProgress brain states, and the most
|
||||
recent assistant text. No LLM call required; this is purely structural
|
||||
extraction from the existing message history.
|
||||
"""
|
||||
if not messages:
|
||||
return ""
|
||||
|
||||
# Find the original user task (first user-text message)
|
||||
initial_task = ""
|
||||
for msg in messages:
|
||||
if msg.get("role") == "user":
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str) and content.strip():
|
||||
initial_task = content.strip()[:300]
|
||||
break
|
||||
|
||||
# Count tool calls by name with key params
|
||||
tool_call_summary: dict[str, list[str]] = {}
|
||||
brain_states: list[str] = []
|
||||
last_assistant_text = ""
|
||||
|
||||
for msg in messages:
|
||||
if msg.get("role") != "assistant":
|
||||
continue
|
||||
content = msg.get("content")
|
||||
if not isinstance(content, list):
|
||||
continue
|
||||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
btype = block.get("type")
|
||||
if btype == "tool_use":
|
||||
name = block.get("name", "unknown")
|
||||
inp = block.get("input") or {}
|
||||
if name == "ReportProgress":
|
||||
# Capture the brain state for inline summary
|
||||
brain_states.append(
|
||||
f" • {inp.get('next_goal', '')[:120]}"
|
||||
)
|
||||
continue
|
||||
# Compact one-line description with key params
|
||||
key_param = ""
|
||||
for k in ("index", "key", "url", "selector", "direction", "text"):
|
||||
if k in inp:
|
||||
v = str(inp[k])[:40]
|
||||
key_param = f"{k}={v}"
|
||||
break
|
||||
desc = f"{name}({key_param})" if key_param else name
|
||||
tool_call_summary.setdefault(name, []).append(desc)
|
||||
elif btype == "text":
|
||||
txt = block.get("text", "").strip()
|
||||
if txt:
|
||||
last_assistant_text = txt
|
||||
|
||||
# Build the summary text
|
||||
parts = ["[Summary of earlier browser-agent activity]"]
|
||||
if initial_task:
|
||||
parts.append(f'Original task: "{initial_task}"')
|
||||
if tool_call_summary:
|
||||
total = sum(len(v) for v in tool_call_summary.values())
|
||||
parts.append(f"Actions taken ({total} total):")
|
||||
# Show count + a couple of representative examples per tool
|
||||
for name in sorted(tool_call_summary.keys()):
|
||||
calls = tool_call_summary[name]
|
||||
count = len(calls)
|
||||
sample = calls[-1] # most recent example
|
||||
if count == 1:
|
||||
parts.append(f" - {sample}")
|
||||
else:
|
||||
parts.append(f" - {sample} (×{count})")
|
||||
if brain_states:
|
||||
parts.append("Recent intents:")
|
||||
parts.extend(brain_states[-5:]) # last 5 brain states
|
||||
if last_assistant_text:
|
||||
snippet = last_assistant_text[:400]
|
||||
parts.append(f"Last update from assistant: {snippet}")
|
||||
parts.append(
|
||||
"(Earlier turn-by-turn details have been compacted to keep the "
|
||||
"context window manageable. Continue from where you left off.)"
|
||||
)
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _trim_history_by_turns(messages: list[dict], max_messages: int) -> list[dict]:
|
||||
"""Compact message history when it exceeds max_messages.
|
||||
|
||||
The Anthropic API requires every `tool_result` block to reference a
|
||||
`tool_use_id` from a previous assistant message. Naive slicing can drop
|
||||
a tool_use while keeping its tool_result, causing 400 errors. This
|
||||
function avoids that by:
|
||||
|
||||
1. Walking forward to find a clean turn boundary (a fresh user-text
|
||||
message that starts a new turn; no tool_result content).
|
||||
2. Summarizing everything BEFORE that boundary into a single user-text
|
||||
message and prepending it to the kept tail.
|
||||
3. If no clean boundary exists at all, returning the original history
|
||||
unchanged. Better to temporarily exceed the cap than to corrupt the
|
||||
conversation and 400 every subsequent request.
|
||||
|
||||
The summary is built programmatically (no LLM call) from the message
|
||||
structure: original task, tool call counts, recent ReportProgress brain
|
||||
states, and last assistant text.
|
||||
"""
|
||||
if len(messages) <= max_messages:
|
||||
return list(messages)
|
||||
|
||||
target_tail_size = max_messages - 1 # leave room for the summary message
|
||||
cut_index: int | None = None
|
||||
|
||||
# First pass: walk forward looking for the EARLIEST clean cut point that
|
||||
# gets us under the cap. This preserves the most recent detail.
|
||||
for i in range(1, len(messages)):
|
||||
if not _is_fresh_user_message(messages[i]):
|
||||
continue
|
||||
if len(messages) - i <= target_tail_size:
|
||||
cut_index = i
|
||||
break
|
||||
|
||||
# Second pass: if no cut point gets us under the cap (e.g. the current
|
||||
# turn alone is bigger than max_messages), use the LATEST clean cut point
|
||||
# available. The tail will still exceed the cap, but it's the smallest
|
||||
# safe history we can produce; and any compaction is better than none.
|
||||
if cut_index is None:
|
||||
for i in range(len(messages) - 1, 0, -1):
|
||||
if _is_fresh_user_message(messages[i]):
|
||||
cut_index = i
|
||||
break
|
||||
|
||||
if cut_index is None:
|
||||
# No clean cut anywhere in the history. Return original; better to
|
||||
# exceed the cap than to corrupt the conversation.
|
||||
return list(messages)
|
||||
|
||||
# Compact: summarize messages[0..cut_index-1], prepend as a single
|
||||
# user-text message, then keep messages[cut_index..end] verbatim.
|
||||
summary_text = _summarize_messages(messages[:cut_index])
|
||||
summary_msg = {"role": "user", "content": summary_text}
|
||||
return [summary_msg] + list(messages[cut_index:])
|
||||
|
||||
BROWSER_TOOLS_SCHEMA = [
|
||||
{
|
||||
"name": "ReportProgress",
|
||||
"description": (
|
||||
"Record your assessment of the previous action and your plan for the "
|
||||
"next one. You MUST call this BEFORE any browser action tools in every "
|
||||
"turn (after the very first turn). This is how you reflect on what just "
|
||||
"happened, track what you've learned about this site, and articulate what "
|
||||
"you're trying to do next. Skipping it is not allowed and will be rejected."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"evaluation_previous": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"What did the previous action(s) accomplish? Did they succeed? "
|
||||
"If not, why? Be specific about what changed on the page."
|
||||
),
|
||||
},
|
||||
"working_memory": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"Short notes about what you've learned about this site so far; "
|
||||
"selectors that work, keyboard shortcuts, layout quirks, what "
|
||||
"you've tried that failed. Carry this forward across turns."
|
||||
),
|
||||
},
|
||||
"next_goal": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"What you're trying to achieve with the action(s) you're about "
|
||||
"to take next. Be concrete."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["evaluation_previous", "working_memory", "next_goal"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserScreenshot",
|
||||
"description": (
|
||||
"Capture a screenshot of the browser page. Returns the screenshot as a "
|
||||
"base64-encoded PNG image. Use this to see what is currently displayed."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserGetText",
|
||||
"description": (
|
||||
"Get the visible text content of the browser page. Returns up to 15000 characters."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserNavigate",
|
||||
"description": "Navigate the browser to a URL.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"url": {"type": "string", "description": "The URL to navigate to."},
|
||||
},
|
||||
"required": ["url"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserClick",
|
||||
"description": "Click an element identified by a CSS selector. Use BrowserGetElements first to discover valid selectors.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"selector": {"type": "string", "description": "CSS selector of the element to click."},
|
||||
},
|
||||
"required": ["selector"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserType",
|
||||
"description": "Type text into an input element. Clears existing value first.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"selector": {"type": "string", "description": "CSS selector of the input element."},
|
||||
"text": {"type": "string", "description": "The text to type."},
|
||||
},
|
||||
"required": ["selector", "text"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserEvaluate",
|
||||
"description": "Evaluate a JavaScript expression in the browser page and return the result.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"expression": {"type": "string", "description": "JavaScript expression to evaluate."},
|
||||
},
|
||||
"required": ["expression"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserGetElements",
|
||||
"description": (
|
||||
"Get a list of interactive elements on the page with CSS selectors. "
|
||||
"Call this BEFORE clicking or typing so you know which selectors are valid."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"selector": {
|
||||
"type": "string",
|
||||
"description": "Optional CSS selector to scope the search (e.g. 'form', '#main'). Defaults to 'body'.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserScroll",
|
||||
"description": (
|
||||
"Scroll the page up or down. Automatically finds the correct scrollable "
|
||||
"container (works on SPAs like Notion, Gmail, etc. that use nested scroll "
|
||||
"containers instead of window-level scrolling). Returns scroll position info "
|
||||
"including whether top/bottom has been reached."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"direction": {
|
||||
"type": "string",
|
||||
"enum": ["up", "down"],
|
||||
"description": "Scroll direction. Defaults to 'down'.",
|
||||
},
|
||||
"amount": {
|
||||
"type": "number",
|
||||
"description": "Pixels to scroll. Defaults to 500.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserListInteractives",
|
||||
"description": (
|
||||
"Get a NUMBERED LIST of interactive elements on the page using the "
|
||||
"browser's accessibility tree. Returns elements like [1]<button \"Like\">, "
|
||||
"[2]<link \"Settings\">, etc. Use this BEFORE BrowserClickIndex. This is "
|
||||
"the PREFERRED way to discover clickable elements on hostile sites "
|
||||
"(Tinder, Instagram, TikTok) where CSS selectors fail because the page "
|
||||
"uses unlabeled <div>s; the accessibility tree sees roles and names "
|
||||
"even when raw HTML doesn't expose them. Much more reliable than "
|
||||
"BrowserGetElements (which uses CSS selectors)."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserClickIndex",
|
||||
"description": (
|
||||
"Click an element by its numeric index from BrowserListInteractives. "
|
||||
"Uses native OS-level mouse events (event.isTrusted=true) so it works "
|
||||
"on sites that filter out synthetic JS events. Always call "
|
||||
"BrowserListInteractives first to get a fresh index list. If the click "
|
||||
"returns 'index no longer valid', the page changed; re-list and retry."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"index": {
|
||||
"type": "integer",
|
||||
"description": "The numeric index from BrowserListInteractives (1-based).",
|
||||
},
|
||||
},
|
||||
"required": ["index"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserBatch",
|
||||
"description": (
|
||||
"Run a sequence of browser actions in one tool call. Each sub-action "
|
||||
"is executed in order, with the URL captured before/after each one. "
|
||||
"If the URL changes mid-batch (the page navigated), the rest of the "
|
||||
"batch is aborted and you get a partial result. Use this when you "
|
||||
"have a known sequence; typing then pressing Enter, swiping multiple "
|
||||
"times, clicking through pagination. Max 5 actions per batch.\n\n"
|
||||
"Sub-action types and their params:\n"
|
||||
"- click_index: { index: int }\n"
|
||||
"- press_key: { key: str }\n"
|
||||
"- type: { selector: str, text: str }\n"
|
||||
"- click: { selector: str }\n"
|
||||
"- scroll: { direction?: 'up'|'down', amount?: int }\n"
|
||||
"- wait: { milliseconds?: int }\n"
|
||||
"- navigate: { url: str }\n\n"
|
||||
"Example: { actions: [{type: 'click_index', params: {index: 1}}, "
|
||||
"{type: 'wait', params: {milliseconds: 500}}, "
|
||||
"{type: 'press_key', params: {key: 'ArrowRight'}}] }"
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"actions": {
|
||||
"type": "array",
|
||||
"maxItems": 5,
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"type": {
|
||||
"type": "string",
|
||||
"enum": ["click_index", "press_key", "type", "wait", "scroll", "navigate", "click"],
|
||||
},
|
||||
"params": {"type": "object"},
|
||||
},
|
||||
"required": ["type", "params"],
|
||||
},
|
||||
},
|
||||
},
|
||||
"required": ["actions"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserPressKey",
|
||||
"description": (
|
||||
"Press a keyboard key (or key combination) on the page using a real native "
|
||||
"input event. Use this for keyboard shortcuts when JS-dispatched events get "
|
||||
"ignored; sites like Tinder, Slack, Notion, Gmail listen for trusted key "
|
||||
"events. Examples: 'ArrowLeft', 'ArrowRight', 'Enter', 'Escape', 'Tab', "
|
||||
"'Space', single letters like 'a'. Prefer this over BrowserEvaluate with "
|
||||
"dispatchEvent for keyboard shortcuts."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"key": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"The key to press. Use JS KeyboardEvent.key names like "
|
||||
"'ArrowUp', 'ArrowDown', 'Enter', 'Escape', 'Tab', 'Space', "
|
||||
"'Backspace', or a single character like 'a'."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["key"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserWait",
|
||||
"description": (
|
||||
"Wait for a specified duration. Useful after navigation or actions that "
|
||||
"trigger page loads, animations, or async content rendering. "
|
||||
"Min 100ms, max 10000ms."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"milliseconds": {
|
||||
"type": "number",
|
||||
"description": "Duration to wait in milliseconds. Defaults to 1000.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "RequestHumanIntervention",
|
||||
"description": (
|
||||
"Request the user's help when you encounter an obstacle you cannot solve "
|
||||
"programmatically; captchas, login prompts, cookie consent walls, "
|
||||
"two-factor authentication, or any blocking popup. The agent will pause "
|
||||
"until the user resolves the issue and clicks Continue."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"problem": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"One short sentence describing the obstacle. Keep it under "
|
||||
"15 words. Example: 'Login required; please sign in to X/Twitter.'"
|
||||
),
|
||||
},
|
||||
"instruction": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"One short sentence telling the user what to do. Keep it under "
|
||||
"15 words. Example: 'Log in with your credentials, then click Done.'"
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["problem", "instruction"],
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
ACTION_MAP = {
|
||||
"BrowserScreenshot": "screenshot",
|
||||
"BrowserGetText": "get_text",
|
||||
"BrowserNavigate": "navigate",
|
||||
"BrowserClick": "click",
|
||||
"BrowserType": "type",
|
||||
"BrowserEvaluate": "evaluate",
|
||||
"BrowserGetElements": "get_elements",
|
||||
"BrowserScroll": "scroll",
|
||||
"BrowserWait": "wait",
|
||||
"BrowserPressKey": "press_key",
|
||||
"BrowserListInteractives": "list_interactives",
|
||||
"BrowserClickIndex": "click_index",
|
||||
"BrowserBatch": "batch",
|
||||
}
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You are a website-agnostic browser automation agent. You can operate on ANY "
|
||||
"website the user is signed into; social media, dating apps, email, productivity "
|
||||
"tools, dashboards, ecommerce, anything. Assume the user has already logged in.\n\n"
|
||||
|
||||
"## Required output structure: ReportProgress before every action\n"
|
||||
"Before ANY action tool (BrowserClick, BrowserType, BrowserNavigate, "
|
||||
"BrowserPressKey, BrowserScroll, BrowserEvaluate, BrowserClickIndex, "
|
||||
"BrowserBatch), you MUST call the ReportProgress tool in the SAME turn. "
|
||||
"ReportProgress takes three short fields:\n"
|
||||
"- evaluation_previous: did your last action work? what changed on the page?\n"
|
||||
"- working_memory: what have you learned about this site? what worked, what didn't?\n"
|
||||
"- next_goal: what specifically are you trying to do with the next action?\n"
|
||||
"Emit ReportProgress and your action tool(s) together in the same response. "
|
||||
"If you skip ReportProgress, your action tools will be REJECTED with an error "
|
||||
"and you will have to retry. This is not optional. Read-only tools "
|
||||
"(BrowserScreenshot, BrowserGetText, BrowserGetElements, BrowserWait) do not "
|
||||
"require ReportProgress.\n\n"
|
||||
|
||||
"## Loop awareness\n"
|
||||
"If you see a tool result containing 'LOOP DETECTED' or '⚠️', it means you "
|
||||
"have called the same tool with the same parameters and gotten the same "
|
||||
"result multiple times in a row. STOP. Do NOT retry the same approach. "
|
||||
"Switch strategy entirely: try a different tool, a different selector, "
|
||||
"keyboard shortcuts, or call RequestHumanIntervention if you genuinely "
|
||||
"cannot proceed. The loop detector will force-exit the agent if you "
|
||||
"ignore it more than 5 times.\n\n"
|
||||
|
||||
"## Use prior context\n"
|
||||
"If this is a continuation of an earlier conversation on the same browser, the "
|
||||
"messages above already contain everything you've tried, what worked, what failed, "
|
||||
"and the page state. READ THAT HISTORY before acting. Do NOT take a fresh screenshot "
|
||||
"or re-explore the DOM if you already know what's on screen; just act. Only re-orient "
|
||||
"if the page has clearly changed (after navigation, after a multi-second wait, or if "
|
||||
"your last action mutated the page in unexpected ways).\n\n"
|
||||
|
||||
"## Try multiple strategies, learn from failures\n"
|
||||
"Sites vary wildly. When one approach fails, switch tactics; don't retry the same "
|
||||
"thing. The escalation ladder, fastest to slowest:\n"
|
||||
"1. **Keyboard shortcuts via BrowserPressKey**; fastest and most reliable on sites "
|
||||
"that support them (Tinder swipes, Gmail navigation, Slack message jump, etc.). "
|
||||
"Always check if the site shows keyboard hints in the UI before falling back to clicks. "
|
||||
"BrowserPressKey sends real native events that pass the `event.isTrusted` check, so "
|
||||
"it works where dispatchEvent in BrowserEvaluate silently fails.\n"
|
||||
"2. **Accessibility tree via BrowserListInteractives + BrowserClickIndex**; the "
|
||||
"accessibility tree sees roles and names that the raw DOM doesn't, even on sites "
|
||||
"like Tinder, Instagram, and TikTok that use unlabeled <div>s with click handlers. "
|
||||
"Call BrowserListInteractives to get a numbered list (`[1]<button \"Like\">`, "
|
||||
"`[2]<link \"Settings\">`), then BrowserClickIndex with the number. The click uses "
|
||||
"native OS-level mouse events so it works where DOM .click() doesn't. THIS IS YOUR "
|
||||
"GO-TO STRATEGY for unlabeled or hostile sites; try this BEFORE BrowserGetElements.\n"
|
||||
"3. **Semantic CSS selectors**; `button[aria-label='X']`, `[role='button']`, "
|
||||
"`a[href*='...']`. Try these via BrowserGetElements + BrowserClick when the site "
|
||||
"actually has semantic HTML.\n"
|
||||
"4. **Text-based JS query**; when both of the above fail, use BrowserEvaluate to "
|
||||
"find elements by visible text: `Array.from(document.querySelectorAll('*')).find(el => el.textContent.trim() === 'Like')`.\n"
|
||||
"5. **Coordinate-based fallback**; last resort: take a screenshot, identify the "
|
||||
"button visually, then click by approximate coords.\n\n"
|
||||
|
||||
"## Batch known sequences with BrowserBatch\n"
|
||||
"When you have a known sequence of actions; typing then pressing Enter, "
|
||||
"swiping multiple times, clicking through pagination; emit them all in a "
|
||||
"single BrowserBatch call instead of one tool per turn. The batch executes "
|
||||
"sub-actions sequentially and aborts if the URL changes mid-batch (so you "
|
||||
"won't operate on stale state). Max 5 sub-actions per batch.\n"
|
||||
"Use BrowserBatch when:\n"
|
||||
"- You're doing the same action repeatedly (5 swipes, 3 scrolls)\n"
|
||||
"- You have a deterministic flow (type query → press Enter → click first result)\n"
|
||||
"Don't use BrowserBatch when:\n"
|
||||
"- You need to read the page state between actions\n"
|
||||
"- You're uncertain about what comes next\n"
|
||||
"- An action might trigger an unexpected popup or navigation\n\n"
|
||||
|
||||
"## Avoid wasted cycles\n"
|
||||
"- Do NOT screenshot after every single action. Screenshot ONLY when you genuinely "
|
||||
"don't know the page state (start of task, after navigation, after a failure).\n"
|
||||
"- Do NOT call BrowserGetElements on the entire body if you already know roughly "
|
||||
"where the target is. Scope it: `BrowserGetElements({selector: 'nav'})`.\n"
|
||||
"- Do NOT call the same failing tool twice with identical parameters. If selector "
|
||||
"X failed, try a DIFFERENT selector or a DIFFERENT strategy.\n"
|
||||
"- For repeated actions (swiping through profiles, going through inbox messages), "
|
||||
"use BrowserPressKey if available; it's an order of magnitude faster than DOM clicks.\n\n"
|
||||
|
||||
"## When you genuinely cannot proceed\n"
|
||||
"Use RequestHumanIntervention for:\n"
|
||||
"- Login walls (the user thinks they're logged in but the session expired)\n"
|
||||
"- Captchas, 2FA prompts, age verification gates\n"
|
||||
"- Anything genuinely ambiguous about user intent\n"
|
||||
"Don't use it for normal tool failures; try a different approach first.\n\n"
|
||||
|
||||
"## Tool reference\n"
|
||||
"- BrowserScreenshot: visual snapshot. Use sparingly, not after every action.\n"
|
||||
"- BrowserGetText: returns up to 15000 chars of visible text. Useful for reading "
|
||||
"content without an image.\n"
|
||||
"- BrowserScroll: handles nested scroll containers (Notion, Gmail). Returns "
|
||||
"atTop/atBottom; stop looping when scroll delta is 0.\n"
|
||||
"- BrowserGetElements: enumerate interactive elements with selectors.\n"
|
||||
"- BrowserClick / BrowserType: standard DOM interaction.\n"
|
||||
"- BrowserPressKey: native key events (preferred for shortcuts).\n"
|
||||
"- BrowserEvaluate: arbitrary JS for everything else, including text-based element "
|
||||
"search and reading state. Avoid for scrolling and keyboard events.\n"
|
||||
"- BrowserWait: 1-3s after navigation, 0.5s after most clicks.\n\n"
|
||||
|
||||
"Complete the task autonomously and report a clear, brief summary."
|
||||
)
|
||||
|
||||
MAX_TURNS = 40
|
||||
|
||||
# Tools that count as "action tools"; calling any of these in a turn requires
|
||||
# the model to also call ReportProgress in the same turn (after the first
|
||||
# turn). Read-only tools and meta tools are exempt.
|
||||
_ACTION_TOOLS_REQUIRING_REPORT = {
|
||||
"BrowserClick",
|
||||
"BrowserType",
|
||||
"BrowserNavigate",
|
||||
"BrowserPressKey",
|
||||
"BrowserScroll",
|
||||
"BrowserEvaluate",
|
||||
"BrowserClickIndex", # Phase 3
|
||||
"BrowserBatch", # Phase 4
|
||||
}
|
||||
|
||||
|
||||
async def execute_browser_tool(
|
||||
tool_name: str, tool_input: dict, browser_id: str, tab_id: str = "",
|
||||
@@ -955,13 +255,13 @@ async def run_browser_agent(
|
||||
# cycle every time the parent issues a new task. Defensively validate
|
||||
# the cache; if it's somehow corrupted (orphaned tool_use_ids), drop
|
||||
# it and start fresh rather than crash on the next API call.
|
||||
prior_messages = _browser_history.get(browser_id) or []
|
||||
prior_messages = browser_history._browser_history.get(browser_id) or []
|
||||
if prior_messages and not _validate_message_pairing(prior_messages):
|
||||
logger.warning(
|
||||
f"[browser-agent {session_id}] cached history for {browser_id} has "
|
||||
f"orphaned tool_use_ids; dropping cache and starting fresh"
|
||||
)
|
||||
_browser_history.pop(browser_id, None)
|
||||
clear_browser_history(browser_id)
|
||||
prior_messages = []
|
||||
messages: list[dict] = list(prior_messages) + [{"role": "user", "content": task}]
|
||||
action_log: list[dict] = []
|
||||
@@ -1379,7 +679,7 @@ async def run_browser_agent(
|
||||
# _MAX_HISTORY_MESSAGES turns to keep token usage bounded; but
|
||||
# never split a tool_use ↔ tool_result pair across the cut, or the
|
||||
# next API request will 400.
|
||||
_browser_history[browser_id] = _trim_history_by_turns(
|
||||
browser_history._browser_history[browser_id] = _trim_history_by_turns(
|
||||
messages, _MAX_HISTORY_MESSAGES,
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1,209 @@
|
||||
"""
|
||||
Conversation-history cache for the browser sub-agent.
|
||||
|
||||
Caches conversation history per browser_id so successive BrowserAgent calls on
|
||||
the same browser can resume rather than restart from scratch. Without this every
|
||||
"swipe right" / "swipe left" call has to take a new screenshot and re-orient
|
||||
itself, costing 30-60s per action.
|
||||
|
||||
The `_browser_history` mutable cache lives in EXACTLY this module; all reads and
|
||||
writes route through here so there's a single source of truth.
|
||||
"""
|
||||
|
||||
# browser_id -> cached Anthropic message list for resume.
|
||||
_browser_history: dict[str, list[dict]] = {}
|
||||
# Cap history to prevent unbounded growth on long-lived browsers.
|
||||
_MAX_HISTORY_MESSAGES = 30
|
||||
|
||||
|
||||
def clear_browser_history(browser_id: str) -> None:
|
||||
"""Drop cached conversation history for a browser (e.g. when it's closed)."""
|
||||
_browser_history.pop(browser_id, None)
|
||||
|
||||
|
||||
def _validate_message_pairing(messages: list[dict]) -> bool:
|
||||
"""Verify every tool_result references a tool_use_id from a prior assistant
|
||||
message in the same list. Returns False if there's an orphan, which means
|
||||
the cached history would 400 if sent to the API.
|
||||
|
||||
This is the last line of defense against cache corruption; if it ever
|
||||
returns False on a resume, we drop the cache and start fresh rather than
|
||||
crash on the next API call.
|
||||
"""
|
||||
declared_tool_use_ids: set[str] = set()
|
||||
for msg in messages:
|
||||
role = msg.get("role")
|
||||
content = msg.get("content")
|
||||
if role == "assistant" and isinstance(content, list):
|
||||
for block in content:
|
||||
if isinstance(block, dict) and block.get("type") == "tool_use":
|
||||
tu_id = block.get("id")
|
||||
if tu_id:
|
||||
declared_tool_use_ids.add(tu_id)
|
||||
elif role == "user" and isinstance(content, list):
|
||||
for block in content:
|
||||
if isinstance(block, dict) and block.get("type") == "tool_result":
|
||||
tr_id = block.get("tool_use_id")
|
||||
if tr_id and tr_id not in declared_tool_use_ids:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _is_fresh_user_message(msg: dict) -> bool:
|
||||
"""A 'fresh' user message starts a new turn; string content or a list
|
||||
that contains no tool_result blocks. These are the only safe cut points
|
||||
because they don't reference any prior assistant tool_use blocks."""
|
||||
if msg.get("role") != "user":
|
||||
return False
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str):
|
||||
return True
|
||||
if isinstance(content, list) and not any(
|
||||
isinstance(c, dict) and c.get("type") == "tool_result" for c in content
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _summarize_messages(messages: list[dict]) -> str:
|
||||
"""Build a programmatic summary of older browser-agent messages.
|
||||
|
||||
Extracts the original user task, a count of tool calls by name with their
|
||||
key parameters, the last few ReportProgress brain states, and the most
|
||||
recent assistant text. No LLM call required; this is purely structural
|
||||
extraction from the existing message history.
|
||||
"""
|
||||
if not messages:
|
||||
return ""
|
||||
|
||||
# Find the original user task (first user-text message)
|
||||
initial_task = ""
|
||||
for msg in messages:
|
||||
if msg.get("role") == "user":
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str) and content.strip():
|
||||
initial_task = content.strip()[:300]
|
||||
break
|
||||
|
||||
# Count tool calls by name with key params
|
||||
tool_call_summary: dict[str, list[str]] = {}
|
||||
brain_states: list[str] = []
|
||||
last_assistant_text = ""
|
||||
|
||||
for msg in messages:
|
||||
if msg.get("role") != "assistant":
|
||||
continue
|
||||
content = msg.get("content")
|
||||
if not isinstance(content, list):
|
||||
continue
|
||||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
btype = block.get("type")
|
||||
if btype == "tool_use":
|
||||
name = block.get("name", "unknown")
|
||||
inp = block.get("input") or {}
|
||||
if name == "ReportProgress":
|
||||
# Capture the brain state for inline summary
|
||||
brain_states.append(
|
||||
f" • {inp.get('next_goal', '')[:120]}"
|
||||
)
|
||||
continue
|
||||
# Compact one-line description with key params
|
||||
key_param = ""
|
||||
for k in ("index", "key", "url", "selector", "direction", "text"):
|
||||
if k in inp:
|
||||
v = str(inp[k])[:40]
|
||||
key_param = f"{k}={v}"
|
||||
break
|
||||
desc = f"{name}({key_param})" if key_param else name
|
||||
tool_call_summary.setdefault(name, []).append(desc)
|
||||
elif btype == "text":
|
||||
txt = block.get("text", "").strip()
|
||||
if txt:
|
||||
last_assistant_text = txt
|
||||
|
||||
# Build the summary text
|
||||
parts = ["[Summary of earlier browser-agent activity]"]
|
||||
if initial_task:
|
||||
parts.append(f'Original task: "{initial_task}"')
|
||||
if tool_call_summary:
|
||||
total = sum(len(v) for v in tool_call_summary.values())
|
||||
parts.append(f"Actions taken ({total} total):")
|
||||
# Show count + a couple of representative examples per tool
|
||||
for name in sorted(tool_call_summary.keys()):
|
||||
calls = tool_call_summary[name]
|
||||
count = len(calls)
|
||||
sample = calls[-1] # most recent example
|
||||
if count == 1:
|
||||
parts.append(f" - {sample}")
|
||||
else:
|
||||
parts.append(f" - {sample} (×{count})")
|
||||
if brain_states:
|
||||
parts.append("Recent intents:")
|
||||
parts.extend(brain_states[-5:]) # last 5 brain states
|
||||
if last_assistant_text:
|
||||
snippet = last_assistant_text[:400]
|
||||
parts.append(f"Last update from assistant: {snippet}")
|
||||
parts.append(
|
||||
"(Earlier turn-by-turn details have been compacted to keep the "
|
||||
"context window manageable. Continue from where you left off.)"
|
||||
)
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _trim_history_by_turns(messages: list[dict], max_messages: int) -> list[dict]:
|
||||
"""Compact message history when it exceeds max_messages.
|
||||
|
||||
The Anthropic API requires every `tool_result` block to reference a
|
||||
`tool_use_id` from a previous assistant message. Naive slicing can drop
|
||||
a tool_use while keeping its tool_result, causing 400 errors. This
|
||||
function avoids that by:
|
||||
|
||||
1. Walking forward to find a clean turn boundary (a fresh user-text
|
||||
message that starts a new turn; no tool_result content).
|
||||
2. Summarizing everything BEFORE that boundary into a single user-text
|
||||
message and prepending it to the kept tail.
|
||||
3. If no clean boundary exists at all, returning the original history
|
||||
unchanged. Better to temporarily exceed the cap than to corrupt the
|
||||
conversation and 400 every subsequent request.
|
||||
|
||||
The summary is built programmatically (no LLM call) from the message
|
||||
structure: original task, tool call counts, recent ReportProgress brain
|
||||
states, and last assistant text.
|
||||
"""
|
||||
if len(messages) <= max_messages:
|
||||
return list(messages)
|
||||
|
||||
target_tail_size = max_messages - 1 # leave room for the summary message
|
||||
cut_index: int | None = None
|
||||
|
||||
# First pass: walk forward looking for the EARLIEST clean cut point that
|
||||
# gets us under the cap. This preserves the most recent detail.
|
||||
for i in range(1, len(messages)):
|
||||
if not _is_fresh_user_message(messages[i]):
|
||||
continue
|
||||
if len(messages) - i <= target_tail_size:
|
||||
cut_index = i
|
||||
break
|
||||
|
||||
# Second pass: if no cut point gets us under the cap (e.g. the current
|
||||
# turn alone is bigger than max_messages), use the LATEST clean cut point
|
||||
# available. The tail will still exceed the cap, but it's the smallest
|
||||
# safe history we can produce; and any compaction is better than none.
|
||||
if cut_index is None:
|
||||
for i in range(len(messages) - 1, 0, -1):
|
||||
if _is_fresh_user_message(messages[i]):
|
||||
cut_index = i
|
||||
break
|
||||
|
||||
if cut_index is None:
|
||||
# No clean cut anywhere in the history. Return original; better to
|
||||
# exceed the cap than to corrupt the conversation.
|
||||
return list(messages)
|
||||
|
||||
# Compact: summarize messages[0..cut_index-1], prepend as a single
|
||||
# user-text message, then keep messages[cut_index..end] verbatim.
|
||||
summary_text = _summarize_messages(messages[:cut_index])
|
||||
summary_msg = {"role": "user", "content": summary_text}
|
||||
return [summary_msg] + list(messages[cut_index:])
|
||||
@@ -0,0 +1,74 @@
|
||||
"""
|
||||
Loop detection for the browser sub-agent.
|
||||
|
||||
Tracks recent state-mutating tool calls in a sliding window. If the model
|
||||
repeats the same (tool, input) with the same result several times, we inject
|
||||
an is_error message in the next tool_result to force a strategy change. This
|
||||
prevents the model from burning the entire turn budget on a failing approach.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
# Tools that are read-only / idempotent and should NOT count toward loop
|
||||
# detection. Repeating these is normal (scrolling through a feed, taking
|
||||
# successive screenshots, polling for an element to appear).
|
||||
_LOOP_DETECTION_EXCLUDED_TOOLS = {
|
||||
"BrowserScreenshot",
|
||||
"BrowserGetText",
|
||||
"BrowserGetElements",
|
||||
"BrowserListInteractives", # Phase 3
|
||||
"BrowserWait",
|
||||
"ReportProgress", # Phase 2
|
||||
"RequestHumanIntervention",
|
||||
}
|
||||
|
||||
_LOOP_WINDOW_SIZE = 5
|
||||
_LOOP_REPEAT_THRESHOLD = 3
|
||||
_LOOP_HARD_CAP = 5
|
||||
|
||||
|
||||
def _hash_tool_call(tool_name: str, tool_input: dict, result: dict) -> tuple[str, str, str]:
|
||||
"""Build a stable hash key for a tool call, including its result.
|
||||
|
||||
Including the result hash means that legitimate progress (same input,
|
||||
different output; e.g. BrowserScroll on a long feed) does NOT count
|
||||
as a loop. Only same-input + same-output is treated as stuck.
|
||||
"""
|
||||
try:
|
||||
input_key = json.dumps(tool_input, sort_keys=True, default=str)
|
||||
except Exception:
|
||||
input_key = repr(tool_input)
|
||||
try:
|
||||
# Truncate the result hash to avoid huge image blobs in the key
|
||||
result_key = json.dumps(result, sort_keys=True, default=str)[:300]
|
||||
except Exception:
|
||||
result_key = repr(result)[:300]
|
||||
return (tool_name, input_key, result_key)
|
||||
|
||||
|
||||
def _detect_loop(
|
||||
recent_calls: list[tuple[str, str, str]],
|
||||
new_call: tuple[str, str, str],
|
||||
) -> bool:
|
||||
"""Return True if `new_call` constitutes a loop given recent history.
|
||||
|
||||
A loop is when the same (tool, input, result) has appeared at least
|
||||
`_LOOP_REPEAT_THRESHOLD` times within the last `_LOOP_WINDOW_SIZE`
|
||||
state-mutating calls (the new call counts as one of those occurrences).
|
||||
"""
|
||||
if new_call[0] in _LOOP_DETECTION_EXCLUDED_TOOLS:
|
||||
return False
|
||||
window = recent_calls[-(_LOOP_WINDOW_SIZE - 1):] + [new_call]
|
||||
matches = sum(1 for c in window if c == new_call)
|
||||
return matches >= _LOOP_REPEAT_THRESHOLD
|
||||
|
||||
|
||||
_LOOP_WARNING_TEXT = (
|
||||
"LOOP DETECTED: You have called this tool with these exact parameters and "
|
||||
"gotten the same result {count} times in a row. STOP retrying this approach "
|
||||
", it is not working. Try a fundamentally different strategy: "
|
||||
"(1) check the page state with BrowserScreenshot or BrowserGetText, "
|
||||
"(2) try a different selector or a different tool, "
|
||||
"(3) use BrowserPressKey for keyboard shortcuts if the site supports them, "
|
||||
"or (4) call RequestHumanIntervention if you genuinely cannot proceed."
|
||||
)
|
||||
@@ -0,0 +1,454 @@
|
||||
"""
|
||||
Static schema + prompt blob for the browser sub-agent.
|
||||
|
||||
One big single-responsibility constant file: tool schema, action map, system
|
||||
prompt, and the turn/report invariants. Exceeds the 300-LOC soft ceiling on
|
||||
purpose because it is one cohesive data blob, not multiple responsibilities.
|
||||
"""
|
||||
|
||||
MODEL_MAP = {
|
||||
"sonnet": "claude-sonnet-4-6",
|
||||
"opus": "claude-opus-4-6",
|
||||
"haiku": "claude-haiku-4-5-20251001",
|
||||
}
|
||||
|
||||
BROWSER_TOOLS_SCHEMA = [
|
||||
{
|
||||
"name": "ReportProgress",
|
||||
"description": (
|
||||
"Record your assessment of the previous action and your plan for the "
|
||||
"next one. You MUST call this BEFORE any browser action tools in every "
|
||||
"turn (after the very first turn). This is how you reflect on what just "
|
||||
"happened, track what you've learned about this site, and articulate what "
|
||||
"you're trying to do next. Skipping it is not allowed and will be rejected."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"evaluation_previous": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"What did the previous action(s) accomplish? Did they succeed? "
|
||||
"If not, why? Be specific about what changed on the page."
|
||||
),
|
||||
},
|
||||
"working_memory": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"Short notes about what you've learned about this site so far; "
|
||||
"selectors that work, keyboard shortcuts, layout quirks, what "
|
||||
"you've tried that failed. Carry this forward across turns."
|
||||
),
|
||||
},
|
||||
"next_goal": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"What you're trying to achieve with the action(s) you're about "
|
||||
"to take next. Be concrete."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["evaluation_previous", "working_memory", "next_goal"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserScreenshot",
|
||||
"description": (
|
||||
"Capture a screenshot of the browser page. Returns the screenshot as a "
|
||||
"base64-encoded PNG image. Use this to see what is currently displayed."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserGetText",
|
||||
"description": (
|
||||
"Get the visible text content of the browser page. Returns up to 15000 characters."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserNavigate",
|
||||
"description": "Navigate the browser to a URL.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"url": {"type": "string", "description": "The URL to navigate to."},
|
||||
},
|
||||
"required": ["url"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserClick",
|
||||
"description": "Click an element identified by a CSS selector. Use BrowserGetElements first to discover valid selectors.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"selector": {"type": "string", "description": "CSS selector of the element to click."},
|
||||
},
|
||||
"required": ["selector"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserType",
|
||||
"description": "Type text into an input element. Clears existing value first.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"selector": {"type": "string", "description": "CSS selector of the input element."},
|
||||
"text": {"type": "string", "description": "The text to type."},
|
||||
},
|
||||
"required": ["selector", "text"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserEvaluate",
|
||||
"description": "Evaluate a JavaScript expression in the browser page and return the result.",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"expression": {"type": "string", "description": "JavaScript expression to evaluate."},
|
||||
},
|
||||
"required": ["expression"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserGetElements",
|
||||
"description": (
|
||||
"Get a list of interactive elements on the page with CSS selectors. "
|
||||
"Call this BEFORE clicking or typing so you know which selectors are valid."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"selector": {
|
||||
"type": "string",
|
||||
"description": "Optional CSS selector to scope the search (e.g. 'form', '#main'). Defaults to 'body'.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserScroll",
|
||||
"description": (
|
||||
"Scroll the page up or down. Automatically finds the correct scrollable "
|
||||
"container (works on SPAs like Notion, Gmail, etc. that use nested scroll "
|
||||
"containers instead of window-level scrolling). Returns scroll position info "
|
||||
"including whether top/bottom has been reached."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"direction": {
|
||||
"type": "string",
|
||||
"enum": ["up", "down"],
|
||||
"description": "Scroll direction. Defaults to 'down'.",
|
||||
},
|
||||
"amount": {
|
||||
"type": "number",
|
||||
"description": "Pixels to scroll. Defaults to 500.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserListInteractives",
|
||||
"description": (
|
||||
"Get a NUMBERED LIST of interactive elements on the page using the "
|
||||
"browser's accessibility tree. Returns elements like [1]<button \"Like\">, "
|
||||
"[2]<link \"Settings\">, etc. Use this BEFORE BrowserClickIndex. This is "
|
||||
"the PREFERRED way to discover clickable elements on hostile sites "
|
||||
"(Tinder, Instagram, TikTok) where CSS selectors fail because the page "
|
||||
"uses unlabeled <div>s; the accessibility tree sees roles and names "
|
||||
"even when raw HTML doesn't expose them. Much more reliable than "
|
||||
"BrowserGetElements (which uses CSS selectors)."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserClickIndex",
|
||||
"description": (
|
||||
"Click an element by its numeric index from BrowserListInteractives. "
|
||||
"Uses native OS-level mouse events (event.isTrusted=true) so it works "
|
||||
"on sites that filter out synthetic JS events. Always call "
|
||||
"BrowserListInteractives first to get a fresh index list. If the click "
|
||||
"returns 'index no longer valid', the page changed; re-list and retry."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"index": {
|
||||
"type": "integer",
|
||||
"description": "The numeric index from BrowserListInteractives (1-based).",
|
||||
},
|
||||
},
|
||||
"required": ["index"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserBatch",
|
||||
"description": (
|
||||
"Run a sequence of browser actions in one tool call. Each sub-action "
|
||||
"is executed in order, with the URL captured before/after each one. "
|
||||
"If the URL changes mid-batch (the page navigated), the rest of the "
|
||||
"batch is aborted and you get a partial result. Use this when you "
|
||||
"have a known sequence; typing then pressing Enter, swiping multiple "
|
||||
"times, clicking through pagination. Max 5 actions per batch.\n\n"
|
||||
"Sub-action types and their params:\n"
|
||||
"- click_index: { index: int }\n"
|
||||
"- press_key: { key: str }\n"
|
||||
"- type: { selector: str, text: str }\n"
|
||||
"- click: { selector: str }\n"
|
||||
"- scroll: { direction?: 'up'|'down', amount?: int }\n"
|
||||
"- wait: { milliseconds?: int }\n"
|
||||
"- navigate: { url: str }\n\n"
|
||||
"Example: { actions: [{type: 'click_index', params: {index: 1}}, "
|
||||
"{type: 'wait', params: {milliseconds: 500}}, "
|
||||
"{type: 'press_key', params: {key: 'ArrowRight'}}] }"
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"actions": {
|
||||
"type": "array",
|
||||
"maxItems": 5,
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"type": {
|
||||
"type": "string",
|
||||
"enum": ["click_index", "press_key", "type", "wait", "scroll", "navigate", "click"],
|
||||
},
|
||||
"params": {"type": "object"},
|
||||
},
|
||||
"required": ["type", "params"],
|
||||
},
|
||||
},
|
||||
},
|
||||
"required": ["actions"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserPressKey",
|
||||
"description": (
|
||||
"Press a keyboard key (or key combination) on the page using a real native "
|
||||
"input event. Use this for keyboard shortcuts when JS-dispatched events get "
|
||||
"ignored; sites like Tinder, Slack, Notion, Gmail listen for trusted key "
|
||||
"events. Examples: 'ArrowLeft', 'ArrowRight', 'Enter', 'Escape', 'Tab', "
|
||||
"'Space', single letters like 'a'. Prefer this over BrowserEvaluate with "
|
||||
"dispatchEvent for keyboard shortcuts."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"key": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"The key to press. Use JS KeyboardEvent.key names like "
|
||||
"'ArrowUp', 'ArrowDown', 'Enter', 'Escape', 'Tab', 'Space', "
|
||||
"'Backspace', or a single character like 'a'."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["key"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserWait",
|
||||
"description": (
|
||||
"Wait for a specified duration. Useful after navigation or actions that "
|
||||
"trigger page loads, animations, or async content rendering. "
|
||||
"Min 100ms, max 10000ms."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"milliseconds": {
|
||||
"type": "number",
|
||||
"description": "Duration to wait in milliseconds. Defaults to 1000.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "RequestHumanIntervention",
|
||||
"description": (
|
||||
"Request the user's help when you encounter an obstacle you cannot solve "
|
||||
"programmatically; captchas, login prompts, cookie consent walls, "
|
||||
"two-factor authentication, or any blocking popup. The agent will pause "
|
||||
"until the user resolves the issue and clicks Continue."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"problem": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"One short sentence describing the obstacle. Keep it under "
|
||||
"15 words. Example: 'Login required; please sign in to X/Twitter.'"
|
||||
),
|
||||
},
|
||||
"instruction": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"One short sentence telling the user what to do. Keep it under "
|
||||
"15 words. Example: 'Log in with your credentials, then click Done.'"
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["problem", "instruction"],
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
ACTION_MAP = {
|
||||
"BrowserScreenshot": "screenshot",
|
||||
"BrowserGetText": "get_text",
|
||||
"BrowserNavigate": "navigate",
|
||||
"BrowserClick": "click",
|
||||
"BrowserType": "type",
|
||||
"BrowserEvaluate": "evaluate",
|
||||
"BrowserGetElements": "get_elements",
|
||||
"BrowserScroll": "scroll",
|
||||
"BrowserWait": "wait",
|
||||
"BrowserPressKey": "press_key",
|
||||
"BrowserListInteractives": "list_interactives",
|
||||
"BrowserClickIndex": "click_index",
|
||||
"BrowserBatch": "batch",
|
||||
}
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You are a website-agnostic browser automation agent. You can operate on ANY "
|
||||
"website the user is signed into; social media, dating apps, email, productivity "
|
||||
"tools, dashboards, ecommerce, anything. Assume the user has already logged in.\n\n"
|
||||
|
||||
"## Required output structure: ReportProgress before every action\n"
|
||||
"Before ANY action tool (BrowserClick, BrowserType, BrowserNavigate, "
|
||||
"BrowserPressKey, BrowserScroll, BrowserEvaluate, BrowserClickIndex, "
|
||||
"BrowserBatch), you MUST call the ReportProgress tool in the SAME turn. "
|
||||
"ReportProgress takes three short fields:\n"
|
||||
"- evaluation_previous: did your last action work? what changed on the page?\n"
|
||||
"- working_memory: what have you learned about this site? what worked, what didn't?\n"
|
||||
"- next_goal: what specifically are you trying to do with the next action?\n"
|
||||
"Emit ReportProgress and your action tool(s) together in the same response. "
|
||||
"If you skip ReportProgress, your action tools will be REJECTED with an error "
|
||||
"and you will have to retry. This is not optional. Read-only tools "
|
||||
"(BrowserScreenshot, BrowserGetText, BrowserGetElements, BrowserWait) do not "
|
||||
"require ReportProgress.\n\n"
|
||||
|
||||
"## Loop awareness\n"
|
||||
"If you see a tool result containing 'LOOP DETECTED' or '⚠️', it means you "
|
||||
"have called the same tool with the same parameters and gotten the same "
|
||||
"result multiple times in a row. STOP. Do NOT retry the same approach. "
|
||||
"Switch strategy entirely: try a different tool, a different selector, "
|
||||
"keyboard shortcuts, or call RequestHumanIntervention if you genuinely "
|
||||
"cannot proceed. The loop detector will force-exit the agent if you "
|
||||
"ignore it more than 5 times.\n\n"
|
||||
|
||||
"## Use prior context\n"
|
||||
"If this is a continuation of an earlier conversation on the same browser, the "
|
||||
"messages above already contain everything you've tried, what worked, what failed, "
|
||||
"and the page state. READ THAT HISTORY before acting. Do NOT take a fresh screenshot "
|
||||
"or re-explore the DOM if you already know what's on screen; just act. Only re-orient "
|
||||
"if the page has clearly changed (after navigation, after a multi-second wait, or if "
|
||||
"your last action mutated the page in unexpected ways).\n\n"
|
||||
|
||||
"## Try multiple strategies, learn from failures\n"
|
||||
"Sites vary wildly. When one approach fails, switch tactics; don't retry the same "
|
||||
"thing. The escalation ladder, fastest to slowest:\n"
|
||||
"1. **Keyboard shortcuts via BrowserPressKey**; fastest and most reliable on sites "
|
||||
"that support them (Tinder swipes, Gmail navigation, Slack message jump, etc.). "
|
||||
"Always check if the site shows keyboard hints in the UI before falling back to clicks. "
|
||||
"BrowserPressKey sends real native events that pass the `event.isTrusted` check, so "
|
||||
"it works where dispatchEvent in BrowserEvaluate silently fails.\n"
|
||||
"2. **Accessibility tree via BrowserListInteractives + BrowserClickIndex**; the "
|
||||
"accessibility tree sees roles and names that the raw DOM doesn't, even on sites "
|
||||
"like Tinder, Instagram, and TikTok that use unlabeled <div>s with click handlers. "
|
||||
"Call BrowserListInteractives to get a numbered list (`[1]<button \"Like\">`, "
|
||||
"`[2]<link \"Settings\">`), then BrowserClickIndex with the number. The click uses "
|
||||
"native OS-level mouse events so it works where DOM .click() doesn't. THIS IS YOUR "
|
||||
"GO-TO STRATEGY for unlabeled or hostile sites; try this BEFORE BrowserGetElements.\n"
|
||||
"3. **Semantic CSS selectors**; `button[aria-label='X']`, `[role='button']`, "
|
||||
"`a[href*='...']`. Try these via BrowserGetElements + BrowserClick when the site "
|
||||
"actually has semantic HTML.\n"
|
||||
"4. **Text-based JS query**; when both of the above fail, use BrowserEvaluate to "
|
||||
"find elements by visible text: `Array.from(document.querySelectorAll('*')).find(el => el.textContent.trim() === 'Like')`.\n"
|
||||
"5. **Coordinate-based fallback**; last resort: take a screenshot, identify the "
|
||||
"button visually, then click by approximate coords.\n\n"
|
||||
|
||||
"## Batch known sequences with BrowserBatch\n"
|
||||
"When you have a known sequence of actions; typing then pressing Enter, "
|
||||
"swiping multiple times, clicking through pagination; emit them all in a "
|
||||
"single BrowserBatch call instead of one tool per turn. The batch executes "
|
||||
"sub-actions sequentially and aborts if the URL changes mid-batch (so you "
|
||||
"won't operate on stale state). Max 5 sub-actions per batch.\n"
|
||||
"Use BrowserBatch when:\n"
|
||||
"- You're doing the same action repeatedly (5 swipes, 3 scrolls)\n"
|
||||
"- You have a deterministic flow (type query → press Enter → click first result)\n"
|
||||
"Don't use BrowserBatch when:\n"
|
||||
"- You need to read the page state between actions\n"
|
||||
"- You're uncertain about what comes next\n"
|
||||
"- An action might trigger an unexpected popup or navigation\n\n"
|
||||
|
||||
"## Avoid wasted cycles\n"
|
||||
"- Do NOT screenshot after every single action. Screenshot ONLY when you genuinely "
|
||||
"don't know the page state (start of task, after navigation, after a failure).\n"
|
||||
"- Do NOT call BrowserGetElements on the entire body if you already know roughly "
|
||||
"where the target is. Scope it: `BrowserGetElements({selector: 'nav'})`.\n"
|
||||
"- Do NOT call the same failing tool twice with identical parameters. If selector "
|
||||
"X failed, try a DIFFERENT selector or a DIFFERENT strategy.\n"
|
||||
"- For repeated actions (swiping through profiles, going through inbox messages), "
|
||||
"use BrowserPressKey if available; it's an order of magnitude faster than DOM clicks.\n\n"
|
||||
|
||||
"## When you genuinely cannot proceed\n"
|
||||
"Use RequestHumanIntervention for:\n"
|
||||
"- Login walls (the user thinks they're logged in but the session expired)\n"
|
||||
"- Captchas, 2FA prompts, age verification gates\n"
|
||||
"- Anything genuinely ambiguous about user intent\n"
|
||||
"Don't use it for normal tool failures; try a different approach first.\n\n"
|
||||
|
||||
"## Tool reference\n"
|
||||
"- BrowserScreenshot: visual snapshot. Use sparingly, not after every action.\n"
|
||||
"- BrowserGetText: returns up to 15000 chars of visible text. Useful for reading "
|
||||
"content without an image.\n"
|
||||
"- BrowserScroll: handles nested scroll containers (Notion, Gmail). Returns "
|
||||
"atTop/atBottom; stop looping when scroll delta is 0.\n"
|
||||
"- BrowserGetElements: enumerate interactive elements with selectors.\n"
|
||||
"- BrowserClick / BrowserType: standard DOM interaction.\n"
|
||||
"- BrowserPressKey: native key events (preferred for shortcuts).\n"
|
||||
"- BrowserEvaluate: arbitrary JS for everything else, including text-based element "
|
||||
"search and reading state. Avoid for scrolling and keyboard events.\n"
|
||||
"- BrowserWait: 1-3s after navigation, 0.5s after most clicks.\n\n"
|
||||
|
||||
"Complete the task autonomously and report a clear, brief summary."
|
||||
)
|
||||
|
||||
MAX_TURNS = 40
|
||||
|
||||
# Tools that count as "action tools"; calling any of these in a turn requires
|
||||
# the model to also call ReportProgress in the same turn (after the first
|
||||
# turn). Read-only tools and meta tools are exempt.
|
||||
_ACTION_TOOLS_REQUIRING_REPORT = {
|
||||
"BrowserClick",
|
||||
"BrowserType",
|
||||
"BrowserNavigate",
|
||||
"BrowserPressKey",
|
||||
"BrowserScroll",
|
||||
"BrowserEvaluate",
|
||||
"BrowserClickIndex", # Phase 3
|
||||
"BrowserBatch", # Phase 4
|
||||
}
|
||||
Reference in New Issue
Block a user