mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-28 04:24:51 +02:00
[pierre] feat: merge app-use (OPENSWARM_APP bridge + AppAgent) onto eric/dev, convention-clean
Brings the app-use feature (OPENSWARM_APP bridge, AppAgent delegation tool, BrowserClickPoint, autopilot supervision) onto the linted eric/dev base. - Resolved 9 conflicts favoring eric's refactored structure; renamed all of pierre's _-prefixed names to p_/P_, kept eric's prior_messages resume guard. - Made the 4 test-exercised bridge helpers public (p-private rule). - Wired AppAgent into browser_delegation_tools + BuiltinTool registry and threaded selected_app_output_ids -> OPENSWARM_SELECTED_APP_IDS (eric had forward-ported the prompt but not the tool, so it was unreachable). - Linter clean (underscore/p-private/ruff/cycles 0); 301 backend tests pass.
This commit is contained in:
@@ -9,6 +9,7 @@ Sub-agents appear as visible AgentSession cards on the dashboard.
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime
|
||||
@@ -69,6 +70,9 @@ from backend.apps.agents.browser import browser_schema
|
||||
from backend.apps.agents.browser.browser_schema import (
|
||||
ACTION_TOOLS_REQUIRING_REPORT,
|
||||
ACTION_MAP,
|
||||
APP_BRIDGE_TOOLS,
|
||||
APP_SYSTEM_PROMPT,
|
||||
APP_VISIBLE_TOOLS,
|
||||
BROWSER_TOOLS_SCHEMA,
|
||||
MAX_TURNS,
|
||||
MODEL_MAP,
|
||||
@@ -87,19 +91,266 @@ P_CONFIRM_TOOLS = {
|
||||
}
|
||||
|
||||
|
||||
def app_bridge_expression(tool_name: str, tool_input: dict) -> str:
|
||||
"""JS for an app bridge tool. Each expression returns a JSON STRING (so it
|
||||
round-trips as text) and never throws; bridge errors come back as JSON."""
|
||||
if tool_name == "AppDescribe":
|
||||
call = "window.OPENSWARM_APP.describe()"
|
||||
elif tool_name == "AppGetState":
|
||||
call = "window.OPENSWARM_APP.getState()"
|
||||
else: # AppInvoke
|
||||
name = json.dumps(tool_input.get("name", ""))
|
||||
args = json.dumps(tool_input.get("args") or {})
|
||||
call = f"window.OPENSWARM_APP.invoke({name}, {args})"
|
||||
return (
|
||||
"(function(){try{"
|
||||
"var A=window.OPENSWARM_APP;"
|
||||
"if(!A||typeof A.describe!=='function'){return JSON.stringify(null);}"
|
||||
f"var r={call};"
|
||||
"return JSON.stringify(r===undefined?null:r);"
|
||||
"}catch(e){return JSON.stringify({__error__:String((e&&e.message)||e)});}})()"
|
||||
)
|
||||
|
||||
|
||||
# App-bridge readiness. The template ships window.OPENSWARM_APP from first paint
|
||||
# but in a "not ready" state until the app calls register(...). On the agent's
|
||||
# first turn the app may still be mounting (Vite cold-boot is 10-30s), so the
|
||||
# reads poll briefly for the bridge to come up instead of declaring it absent.
|
||||
P_BRIDGE_READY_WAIT_MS = 8000
|
||||
P_BRIDGE_POLL_INTERVAL_MS = 400
|
||||
p_bridge_known_absent: set[str] = set()
|
||||
|
||||
|
||||
def parse_bridge_result(result: dict) -> object:
|
||||
"""Decode the JSON string an app-bridge evaluate returns (it always returns
|
||||
JSON text and never throws). Returns the decoded value, or None when it is
|
||||
undecodable or errored at the transport level."""
|
||||
if not isinstance(result, dict) or "error" in result:
|
||||
return None
|
||||
raw = result.get("text")
|
||||
if raw is None:
|
||||
return None
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def bridge_ready(value: object) -> bool:
|
||||
"""True when a decoded describe()/getState() value means a registered bridge.
|
||||
Legacy apps return a plain array (ready); the template stub returns
|
||||
{'__ready': False} until register() runs; None means no bridge present yet."""
|
||||
if isinstance(value, list):
|
||||
return True
|
||||
if isinstance(value, dict):
|
||||
return value.get("__ready") is not False and "__error__" not in value
|
||||
return value is not None
|
||||
|
||||
|
||||
def p_app_output_id(browser_id: str) -> str | None:
|
||||
return browser_id[4:] if browser_id.startswith("app:") else None
|
||||
|
||||
|
||||
def p_app_workspace_dir(browser_id: str) -> str | None:
|
||||
"""Resolve the on-disk workspace folder for an `app:<output_id>` target."""
|
||||
oid = p_app_output_id(browser_id)
|
||||
if not oid:
|
||||
return None
|
||||
try:
|
||||
from backend.apps.outputs.workspace_io import load_output
|
||||
from backend.config.paths import OUTPUTS_WORKSPACE_DIR
|
||||
out = load_output(oid)
|
||||
if not out or not getattr(out, "workspace_id", None):
|
||||
return None
|
||||
return os.path.join(OUTPUTS_WORKSPACE_DIR, out.workspace_id)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def render_app_controls(describe_value: object) -> tuple[str, str] | None:
|
||||
"""From a decoded describe() value build (rules_md, controls_md). Returns
|
||||
None when the value is not a ready, usable describe."""
|
||||
if isinstance(describe_value, list):
|
||||
rules, controls = "", describe_value
|
||||
elif bridge_ready(describe_value) and isinstance(describe_value, dict):
|
||||
rules = str(describe_value.get("rules") or "")
|
||||
controls = describe_value.get("controls") or []
|
||||
else:
|
||||
return None
|
||||
lines = ["# Controls", ""]
|
||||
for c in controls:
|
||||
if not isinstance(c, dict):
|
||||
continue
|
||||
row = f"- `{c.get('name', '')}`"
|
||||
if c.get("args"):
|
||||
row += f" args={json.dumps(c['args'])}"
|
||||
if c.get("keys"):
|
||||
row += f" [{c['keys']}]"
|
||||
if c.get("description"):
|
||||
row += f": {c['description']}"
|
||||
lines.append(row)
|
||||
controls_md = "\n".join(lines) + "\n"
|
||||
rules_md = (rules.strip() + "\n") if rules.strip() else ""
|
||||
return rules_md, controls_md
|
||||
|
||||
|
||||
def p_persist_app_controls(browser_id: str, describe_value: object) -> None:
|
||||
"""Cache rules.md + controls.md into the app workspace so the agent reads them
|
||||
up front and only re-describes when controls change. Best-effort; written
|
||||
under .openswarm/ so they stay out of the app's own file tree."""
|
||||
rendered = render_app_controls(describe_value)
|
||||
if not rendered:
|
||||
return
|
||||
folder = p_app_workspace_dir(browser_id)
|
||||
if not folder or not os.path.isdir(folder):
|
||||
return
|
||||
rules_md, controls_md = rendered
|
||||
try:
|
||||
cache_dir = os.path.join(folder, ".openswarm")
|
||||
os.makedirs(cache_dir, exist_ok=True)
|
||||
with open(os.path.join(cache_dir, "controls.md"), "w", encoding="utf-8") as f:
|
||||
f.write(controls_md)
|
||||
if rules_md:
|
||||
with open(os.path.join(cache_dir, "rules.md"), "w", encoding="utf-8") as f:
|
||||
f.write(rules_md)
|
||||
except Exception:
|
||||
logger.debug("[app-agent] failed to persist controls cache", exc_info=True)
|
||||
|
||||
|
||||
# Single-tool names -> the sub-action type they map to, so one summarizer covers
|
||||
# both BrowserPressKey({key}) and a batch's {"type":"press_key","params":{key}}.
|
||||
P_SINGLE_ACTION_TYPE = {
|
||||
"BrowserClick": "click", "BrowserClickIndex": "click_index",
|
||||
"BrowserClickByName": "click_name", "BrowserType": "type",
|
||||
"BrowserPressKey": "press_key", "BrowserScroll": "scroll",
|
||||
"BrowserNavigate": "navigate", "BrowserClickPoint": "click_point",
|
||||
}
|
||||
|
||||
|
||||
def p_summ_step(stype: str, params: dict) -> str:
|
||||
"""Compact human label for one action step: the actual key/selector/text, not
|
||||
just the verb. This is what lets the [backend] pane show 'key:ArrowRight x5'
|
||||
instead of an opaque 'BrowserBatch'."""
|
||||
p = params or {}
|
||||
if stype == "press_key":
|
||||
return f"key:{p.get('key', '?')}"
|
||||
if stype == "click":
|
||||
return f"click({p.get('selector', '?')})"
|
||||
if stype == "click_index":
|
||||
return f"click#{p.get('index', '?')}"
|
||||
if stype == "click_point":
|
||||
return f"tap({p.get('xPercent', '?')}%,{p.get('yPercent', '?')}%)"
|
||||
if stype == "click_name":
|
||||
return f"clickName({p.get('name', '?')})"
|
||||
if stype == "type":
|
||||
return f"type({p.get('selector', '')}={str(p.get('text', ''))[:30]!r})"
|
||||
if stype == "wait":
|
||||
return f"wait({p.get('milliseconds') or p.get('until') or ''})"
|
||||
if stype == "scroll":
|
||||
return f"scroll({p.get('direction', 'down')})"
|
||||
if stype == "navigate":
|
||||
return f"nav({str(p.get('url', ''))[:60]})"
|
||||
if stype == "list_interactives":
|
||||
return "list"
|
||||
return stype or "?"
|
||||
|
||||
|
||||
def p_collapse_steps(items: list[str]) -> str:
|
||||
"""'ArrowRight, ArrowRight, ArrowRight' -> 'key:ArrowRight x3' so a 5-key
|
||||
burst reads as one token instead of scrolling the pane."""
|
||||
runs: list[list] = []
|
||||
for it in items:
|
||||
if runs and runs[-1][0] == it:
|
||||
runs[-1][1] += 1
|
||||
else:
|
||||
runs.append([it, 1])
|
||||
return ", ".join(s if n == 1 else f"{s} x{n}" for s, n in runs)
|
||||
|
||||
|
||||
def p_summarize_action(tool_name: str, tool_input: dict) -> str:
|
||||
"""One-line summary of what an action tool is about to do, or "" for pure
|
||||
reads (screenshot/list/describe/getstate) that need no action log."""
|
||||
ti = tool_input or {}
|
||||
if tool_name == "BrowserBatch":
|
||||
steps = [p_summ_step((a or {}).get("type", ""), (a or {}).get("params"))
|
||||
for a in (ti.get("actions") or [])]
|
||||
return p_collapse_steps(steps) or "(empty batch)"
|
||||
if tool_name == "AppInvoke":
|
||||
args = ti.get("args")
|
||||
return f"{ti.get('name', '?')}" + (f"({json.dumps(args)[:60]})" if args else "")
|
||||
stype = P_SINGLE_ACTION_TYPE.get(tool_name)
|
||||
return p_summ_step(stype, ti) if stype else ""
|
||||
|
||||
|
||||
async def execute_browser_tool(
|
||||
tool_name: str, tool_input: dict, browser_id: str, tab_id: str = "",
|
||||
) -> dict:
|
||||
"""Execute a browser tool via ws_manager directly (no MCP/HTTP round-trip)."""
|
||||
# [app-agent] step trace: only for app targets / bridge tools so normal
|
||||
# browser-agent runs stay quiet. Greppable prefix; remove when done.
|
||||
p_trace = browser_id.startswith("app:") or tool_name in APP_BRIDGE_TOOLS
|
||||
|
||||
# One greppable line naming the actual buttons/keys/selectors this call drives,
|
||||
# so a run reads as "key:ArrowRight x5" rather than an opaque tool name. Fires
|
||||
# for action tools only (reads stay quiet) and ungated so web runs get it too.
|
||||
p_action = p_summarize_action(tool_name, tool_input)
|
||||
if p_action:
|
||||
logger.info(f"[browser-action] {tool_name}: {p_action} -> {browser_id}")
|
||||
|
||||
# App bridge tools translate to a single BrowserEvaluate against the app's
|
||||
# window.OPENSWARM_APP, so they need no frontend command-handler changes.
|
||||
if tool_name in APP_BRIDGE_TOOLS:
|
||||
action = "evaluate"
|
||||
expr = app_bridge_expression(tool_name, tool_input)
|
||||
params = {"expression": expr}
|
||||
|
||||
async def p_eval_once() -> dict:
|
||||
rid = uuid4().hex
|
||||
if p_trace:
|
||||
logger.info(f"[app-agent] DISPATCH {tool_name} -> {browser_id} (req {rid[:8]}) js={expr[:120]}")
|
||||
r = await ws_manager.send_browser_command(rid, action, browser_id, params, tab_id=tab_id)
|
||||
if p_trace:
|
||||
logger.info(f"[app-agent] RESULT {tool_name} <- {browser_id}: {json.dumps(r)[:300]}")
|
||||
return r
|
||||
|
||||
result = await p_eval_once()
|
||||
# Reads poll for the bridge to come up (app still mounting on turn 1).
|
||||
# AppInvoke does not wait: its action either exists right now or it does
|
||||
# not, and a missing action should surface immediately.
|
||||
if tool_name in ("AppDescribe", "AppGetState"):
|
||||
ready = bridge_ready(parse_bridge_result(result))
|
||||
# Only the cold-boot read polls. Once a card is memoed bridge-absent,
|
||||
# every later read takes the single eval above and skips the 8s sink;
|
||||
# that single eval still detects a bridge that came up between reads.
|
||||
if not ready and browser_id not in p_bridge_known_absent:
|
||||
waited = 0
|
||||
while waited < P_BRIDGE_READY_WAIT_MS and not ready:
|
||||
await asyncio.sleep(P_BRIDGE_POLL_INTERVAL_MS / 1000)
|
||||
waited += P_BRIDGE_POLL_INTERVAL_MS
|
||||
result = await p_eval_once()
|
||||
ready = bridge_ready(parse_bridge_result(result))
|
||||
# Record the verdict so the next read knows whether to pay the wait.
|
||||
if ready:
|
||||
p_bridge_known_absent.discard(browser_id)
|
||||
else:
|
||||
p_bridge_known_absent.add(browser_id)
|
||||
if tool_name == "AppDescribe":
|
||||
p_persist_app_controls(browser_id, parse_bridge_result(result))
|
||||
return result
|
||||
|
||||
action = ACTION_MAP.get(tool_name)
|
||||
if not action:
|
||||
return {"error": f"Unknown browser tool: {tool_name}"}
|
||||
|
||||
params = {k: v for k, v in tool_input.items()}
|
||||
request_id = uuid4().hex
|
||||
if p_trace:
|
||||
logger.info(f"[app-agent] DISPATCH {tool_name}/{action} -> {browser_id} (req {request_id[:8]})")
|
||||
result = await ws_manager.send_browser_command(
|
||||
request_id, action, browser_id, params, tab_id=tab_id,
|
||||
)
|
||||
if p_trace:
|
||||
logger.info(f"[app-agent] RESULT {tool_name} <- {browser_id}: {json.dumps(result)[:300]}")
|
||||
return result
|
||||
|
||||
|
||||
@@ -337,27 +588,36 @@ async def run_browser_agent(
|
||||
pre_selected: bool = False,
|
||||
initial_url: str | None = None,
|
||||
parent_session_id: str | None = None,
|
||||
app_mode: bool = False,
|
||||
) -> dict:
|
||||
"""Run a browser sub-agent loop for a single browser card.
|
||||
|
||||
Creates a visible AgentSession, streams progress via WebSocket,
|
||||
and returns the full action log + summary + final screenshot.
|
||||
|
||||
When app_mode is set, browser_id points at an OpenSwarm-built app's webview
|
||||
(registered as "app:<output_id>"). The agent drives it through the app's
|
||||
native bridge (window.OPENSWARM_APP) instead of web perception: no initial
|
||||
navigate, no AX-tree front-load, and a lean app toolset + prompt.
|
||||
"""
|
||||
from backend.apps.agents.agent_manager import agent_manager
|
||||
|
||||
p_browser_perms = load_builtin_permissions()
|
||||
|
||||
if app_mode:
|
||||
logger.info(f"[app-agent] START loop: browser_id={browser_id} task={task[:140]!r}")
|
||||
|
||||
session_id = uuid4().hex
|
||||
cancel_event = asyncio.Event()
|
||||
session = AgentSession(
|
||||
id=session_id,
|
||||
name=f"Browser Agent",
|
||||
name="App Agent" if app_mode else "Browser Agent",
|
||||
model=model,
|
||||
mode="browser-agent",
|
||||
status="running",
|
||||
dashboard_id=dashboard_id,
|
||||
browser_id=browser_id,
|
||||
system_prompt=SYSTEM_PROMPT,
|
||||
system_prompt=APP_SYSTEM_PROMPT if app_mode else SYSTEM_PROMPT,
|
||||
parent_session_id=parent_session_id,
|
||||
)
|
||||
agent_manager.cancel_events[session_id] = cancel_event
|
||||
@@ -411,15 +671,17 @@ async def run_browser_agent(
|
||||
current_url = ""
|
||||
preloaded_reads: list[dict] = [] # real front-loaded reads, seeded into action_log
|
||||
p_resumed = bool(browser_history.BROWSER_HISTORY.get(browser_id))
|
||||
if initial_url:
|
||||
nav_result = await execute_browser_tool(
|
||||
"BrowserNavigate", {"url": initial_url}, browser_id, tab_id,
|
||||
)
|
||||
logger.info(f"Browser agent {session_id}: navigated to {initial_url}: {nav_result.get('text', nav_result.get('error', ''))}")
|
||||
preloaded_perception, current_url, preloaded_reads = await p_perceive(initial_url)
|
||||
elif not p_resumed:
|
||||
# Fresh task on an existing card: perceive the current page to learn its host (for replay) and front-load turn 1 (this path used to start cold).
|
||||
preloaded_perception, current_url, preloaded_reads = await p_perceive("")
|
||||
# App mode skips this whole block: the app is already loaded and its DOM is uninformative (often a bare <canvas>), so the agent perceives via the bridge (AppDescribe) on turn 1 instead of navigating or front-loading the AX tree.
|
||||
if not app_mode:
|
||||
if initial_url:
|
||||
nav_result = await execute_browser_tool(
|
||||
"BrowserNavigate", {"url": initial_url}, browser_id, tab_id,
|
||||
)
|
||||
logger.info(f"Browser agent {session_id}: navigated to {initial_url}: {nav_result.get('text', nav_result.get('error', ''))}")
|
||||
preloaded_perception, current_url, preloaded_reads = await p_perceive(initial_url)
|
||||
elif not p_resumed:
|
||||
# Fresh task on an existing card: perceive the current page to learn its host (for replay) and front-load turn 1 (this path used to start cold).
|
||||
preloaded_perception, current_url, preloaded_reads = await p_perceive("")
|
||||
|
||||
from backend.apps.settings.settings import load_settings
|
||||
from backend.apps.settings.credentials import get_anthropic_client_for_model
|
||||
@@ -473,8 +735,61 @@ async def run_browser_agent(
|
||||
)
|
||||
clear_browser_history(browser_id)
|
||||
prior_messages = []
|
||||
# Front-load the prefetched perception into the first user turn so the model can act immediately (only when this is a fresh conversation; a resumed one already knows the page). The visible task text stays clean.
|
||||
first_user_content = task + preloaded_perception if (preloaded_perception and not prior_messages) else task
|
||||
# App mode: read the bridge's rules + controls ONCE up front and front-load them so the agent knows the app's purpose and every control before its first action (no screenshot fumbling) and need not call AppDescribe again until controls change. Also the runtime bridge gate: if the bridge never comes up, fail loudly into the logs + the agent's first message (and, under OPENSWARM_REQUIRE_BRIDGE=1, end the run rather than UI-fumble).
|
||||
# NOTE: app mode runs this on EVERY task, even a resumed conversation; the bridge can appear between runs (app just made agent-operable) and a resumed history may carry a stale "no bridge, screenshot it" strategy, so re-reading + re-attaching the controls each task re-points the agent at the bridge.
|
||||
app_front_load = ""
|
||||
if app_mode:
|
||||
try:
|
||||
p_dv = parse_bridge_result(await execute_browser_tool("AppDescribe", {}, browser_id, tab_id))
|
||||
except Exception:
|
||||
p_dv = None
|
||||
logger.debug("[app-agent] startup AppDescribe failed", exc_info=True)
|
||||
p_rendered = render_app_controls(p_dv)
|
||||
if p_rendered:
|
||||
p_rules_md, p_controls_md = p_rendered
|
||||
p_rev = p_dv.get("__rev") if isinstance(p_dv, dict) else None
|
||||
p_block = [
|
||||
"\n\n[The app's bridge is live; its rules and controls were read "
|
||||
"for you. Act directly; do NOT call AppDescribe again unless "
|
||||
"AppGetState reports a changed __rev.]"
|
||||
]
|
||||
if p_rules_md.strip():
|
||||
p_block.append("App rules / objective:\n" + p_rules_md.strip())
|
||||
p_block.append(p_controls_md.strip())
|
||||
if p_rev is not None:
|
||||
p_block.append(f"(controls __rev: {p_rev})")
|
||||
app_front_load = "\n\n".join(p_block)
|
||||
else:
|
||||
p_oid = p_app_output_id(browser_id) or browser_id
|
||||
p_msg = (
|
||||
f"BRIDGE MISSING: window.OPENSWARM_APP not registered - "
|
||||
f"app '{p_oid}' is not agent-operable"
|
||||
)
|
||||
logger.error(f"[app-agent] {p_msg}")
|
||||
if os.environ.get("OPENSWARM_REQUIRE_BRIDGE") == "1":
|
||||
session.status = "completed"
|
||||
await ws_manager.send_to_session(session_id, "agent:status", {
|
||||
"session_id": session_id, "status": "completed",
|
||||
"session": session.model_dump(mode="json"),
|
||||
})
|
||||
return {
|
||||
"session_id": session_id, "browser_id": browser_id,
|
||||
"summary": f"This app is not agent-operable: {p_msg}.",
|
||||
"done": True, "success": False,
|
||||
"action_log": [], "final_screenshot": None,
|
||||
}
|
||||
app_front_load = (
|
||||
f"\n\n[{p_msg}. Operate it like a person instead: see with "
|
||||
"BrowserScreenshot, then play with BrowserPressKey (keys like "
|
||||
"w/a/s/d, arrows, Space, Enter) and BrowserClickPoint (tap a screen "
|
||||
"point). For a normal HTML app use BrowserListInteractives + "
|
||||
"BrowserClickIndex. Only give up (Done success=false) after you have "
|
||||
"actually tried pressing keys and nothing responds.]"
|
||||
)
|
||||
|
||||
# Front-load perception (browser) or the app's controls (app mode) into the new task's user turn so the model can act immediately. Browser perception only attaches on a fresh conversation; app-mode controls attach on every task (see note above). The visible task text stays clean.
|
||||
p_front = app_front_load if app_mode else (preloaded_perception if not prior_messages else "")
|
||||
first_user_content = task + p_front if p_front else task
|
||||
messages: list[dict] = list(prior_messages) + [{"role": "user", "content": first_user_content}]
|
||||
# Seed with the front-loaded reads: they really ran and returned content, so a read task the agent answers straight from them is NOT a "did nothing" ghost.
|
||||
action_log: list[dict] = list(preloaded_reads)
|
||||
@@ -546,8 +861,8 @@ async def run_browser_agent(
|
||||
current_next_goal = task
|
||||
|
||||
# Advisory per-domain hints: seed the system prompt with what a prior agent learned about this domain (if we know the domain at start), and keep the store fresh from each ReportProgress. Re-verify, never blindly trust.
|
||||
start_domain = p_extract_domain(initial_url) if initial_url else None
|
||||
run_system_prompt = SYSTEM_PROMPT
|
||||
start_domain = None if app_mode else (p_extract_domain(initial_url) if initial_url else None)
|
||||
run_system_prompt = APP_SYSTEM_PROMPT if app_mode else SYSTEM_PROMPT
|
||||
if start_domain:
|
||||
prior_note = browser_history.get_domain_note(start_domain)
|
||||
if prior_note:
|
||||
@@ -561,26 +876,28 @@ async def run_browser_agent(
|
||||
|
||||
# Tier-2 memory: seed the DURABLE strategy playbook for this host (distilled from past successful runs) so the model skips re-discovery. Advisory text, re-verified by the agent, never auto-run. Keyed by full host like skills.
|
||||
pb_seeded = False # whether tier-2 strategy was injected, for measuring its effect
|
||||
p_pb_host = browser_skills.host_of(initial_url or current_url or "")
|
||||
# Playbook key: web keys by site host; app mode keys by the stable app id (app:<output_id>). A no-bridge canvas app (Paint, Doom) has no URL host, so without this it re-discovers its own geography cold on every run; keying the durable playbook by browser_id lets that knowledge accumulate across runs.
|
||||
p_pb_host = browser_id if app_mode else browser_skills.host_of(initial_url or current_url or "")
|
||||
if p_pb_host:
|
||||
p_pb_block = browser_playbook.format_for_prompt(p_pb_host)
|
||||
if p_pb_block:
|
||||
run_system_prompt = run_system_prompt + p_pb_block
|
||||
pb_seeded = True
|
||||
# Tier-3 memory: the cross-site priors learned on EVERY other site, injected on every run (host-agnostic) so a brand-new site isn't fully cold. Advisory, capped.
|
||||
try:
|
||||
p_meta_block = browser_meta_playbook.format_for_prompt()
|
||||
if p_meta_block:
|
||||
run_system_prompt = run_system_prompt + p_meta_block
|
||||
except Exception:
|
||||
pass
|
||||
# Tier-3 memory: cross-site priors learned on EVERY other site, injected on every web run (host-agnostic) so a brand-new site isn't fully cold. Web-only: generic site heuristics don't transfer to a bespoke app canvas.
|
||||
if not app_mode:
|
||||
try:
|
||||
p_meta_block = browser_meta_playbook.format_for_prompt()
|
||||
if p_meta_block:
|
||||
run_system_prompt = run_system_prompt + p_meta_block
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Prompt-caching shapes built once: system as a single cached text block, and the last tool carrying the cache_control marker (Anthropic keys on the trailing marker, so one marker covers the whole tool array + system).
|
||||
p_cached_system = [{
|
||||
"type": "text", "text": run_system_prompt,
|
||||
"cache_control": {"type": "ephemeral"},
|
||||
}]
|
||||
p_cached_tools = [dict(t) for t in browser_schema.MODEL_VISIBLE_TOOLS]
|
||||
p_cached_tools = [dict(t) for t in (APP_VISIBLE_TOOLS if app_mode else browser_schema.MODEL_VISIBLE_TOOLS)]
|
||||
if p_cached_tools:
|
||||
p_cached_tools[-1] = {**p_cached_tools[-1], "cache_control": {"type": "ephemeral"}}
|
||||
|
||||
@@ -594,8 +911,9 @@ async def run_browser_agent(
|
||||
# Perceived value, zero clicks: one calm line so the user FEELS the agent is picking up where it left off, not figuring the site out cold again. Only when strategy was actually seeded, so it's honest, never noise.
|
||||
if pb_seeded and p_pb_host:
|
||||
session.memory_recalled = True # drives the subtle "Remembered" card chip
|
||||
p_where = "this app" if app_mode else p_pb_host
|
||||
p_recall_msg = Message(role="assistant",
|
||||
content=f"Picking up what I learned about {p_pb_host} from a previous visit.")
|
||||
content=f"Picking up what I learned about {p_where} from a previous visit.")
|
||||
session.messages.append(p_recall_msg)
|
||||
await ws_manager.send_to_session(session_id, "agent:message", {
|
||||
"session_id": session_id, "message": p_recall_msg.model_dump(mode="json"),
|
||||
@@ -1872,18 +2190,20 @@ async def run_browser_agent(
|
||||
# Tier-2 memory: on a substantive verified success, distill this run into the DURABLE strategy playbook (one cheap aux call, mem0-style distill+ reconcile). Fires for BOTH mechanical and judgment tasks, it's how the judgment ones (which can't be skills) still get faster/wiser next time.
|
||||
if browser_playbook.should_learn(honest, turn + 1):
|
||||
try:
|
||||
pb_host = browser_skills.host_of(last_seen_url)
|
||||
if pb_host:
|
||||
# App mode keys by the stable app id (the run had no URL host); web keys by the final host, which navigation may have changed.
|
||||
rec_pb_host = browser_id if app_mode else browser_skills.host_of(last_seen_url)
|
||||
if rec_pb_host:
|
||||
aux_client, aux_model = await p_get_aux_client()
|
||||
changed = await browser_playbook.distill_and_store(
|
||||
pb_host, skill_key_task, latest_working_mem, summary,
|
||||
rec_pb_host, skill_key_task, latest_working_mem, summary,
|
||||
aux_client, aux_model,
|
||||
)
|
||||
# Perceived value, zero clicks: a calm closing line so the user sees the agent got a little smarter for next time. Only when it genuinely learned something, so it stays honest + rare.
|
||||
if changed:
|
||||
session.memory_learned = True # drives the subtle "Learned" card chip
|
||||
p_where = "this app" if app_mode else rec_pb_host
|
||||
p_learn_msg = Message(role="assistant",
|
||||
content=f"Noted what worked on {pb_host} so I'm faster here next time.")
|
||||
content=f"Noted what worked on {p_where} so I'm faster here next time.")
|
||||
session.messages.append(p_learn_msg)
|
||||
await ws_manager.send_to_session(session_id, "agent:message", {
|
||||
"session_id": session_id, "message": p_learn_msg.model_dump(mode="json"),
|
||||
@@ -2058,6 +2378,8 @@ async def run_browser_agents(
|
||||
browser_id = task_def.get("browser_id", "")
|
||||
task_text = task_def.get("task", "")
|
||||
url = task_def.get("url", "")
|
||||
# App mode: browser_id is a pre-registered app webview ("app:<output_id>"); never create/reuse a browser card, never navigate (the app is loaded).
|
||||
app_mode = bool(task_def.get("app_mode"))
|
||||
# advisory deep entry (from the fast-path brief): a NEW card opens on it directly (no google detour); a REUSED card is never moved by it, so a warm card's deeper page state always wins
|
||||
entry_url = task_def.get("entry_url", "")
|
||||
|
||||
@@ -2084,7 +2406,7 @@ async def run_browser_agents(
|
||||
pass
|
||||
else:
|
||||
await asyncio.sleep(2.0)
|
||||
elif browser_id:
|
||||
elif browser_id and not app_mode:
|
||||
ACTIVE_AGENT_CARDS.add(browser_id)
|
||||
|
||||
is_pre_selected = browser_id in pre_selected
|
||||
@@ -2097,11 +2419,13 @@ async def run_browser_agents(
|
||||
dashboard_id=dashboard_id,
|
||||
pre_selected=is_pre_selected,
|
||||
# an explicit url means "go here" even on the user's picked card; with none, a picked card stays on the page they parked it
|
||||
initial_url=p_nav_url if p_nav_url and (url or browser_id not in pre_selected) else None,
|
||||
initial_url=None if app_mode else (p_nav_url if p_nav_url and (url or browser_id not in pre_selected) else None),
|
||||
parent_session_id=parent_session_id,
|
||||
app_mode=app_mode,
|
||||
)
|
||||
finally:
|
||||
ACTIVE_AGENT_CARDS.discard(browser_id)
|
||||
if not app_mode:
|
||||
ACTIVE_AGENT_CARDS.discard(browser_id)
|
||||
|
||||
results = await asyncio.gather(*[p_run_one(t) for t in tasks], return_exceptions=True)
|
||||
|
||||
|
||||
@@ -400,6 +400,8 @@ BROWSER_TOOLS_SCHEMA = [
|
||||
"Sub-action types and their params:\n"
|
||||
"- click_index: { index: int }\n"
|
||||
"- press_key: { key: str }\n"
|
||||
"- click_point: { xPercent: number, yPercent: number, hold_ms?: int } "
|
||||
"(tap a screen point; for canvas apps/games)\n"
|
||||
"- type: { selector: str, text: str }\n"
|
||||
"- click: { selector: str }\n"
|
||||
"- scroll: { direction?: 'up'|'down', amount?: int }\n"
|
||||
@@ -424,7 +426,7 @@ BROWSER_TOOLS_SCHEMA = [
|
||||
"properties": {
|
||||
"type": {
|
||||
"type": "string",
|
||||
"enum": ["click_index", "press_key", "type", "wait", "scroll", "navigate", "click", "list_interactives"],
|
||||
"enum": ["click_index", "press_key", "click_point", "type", "wait", "scroll", "navigate", "click", "list_interactives"],
|
||||
},
|
||||
"params": {"type": "object"},
|
||||
},
|
||||
@@ -460,6 +462,43 @@ BROWSER_TOOLS_SCHEMA = [
|
||||
"required": ["key"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserClickPoint",
|
||||
"description": (
|
||||
"Tap/click at a point on the screen using a real native mouse event, "
|
||||
"WITHOUT needing a DOM element. This is the way to operate a <canvas> "
|
||||
"app or game (the kind with no clickable HTML elements): you click a "
|
||||
"spot the way a person does. Give the position as a PERCENT of the view "
|
||||
"(xPercent/yPercent, 0-100, with 0,0 = top-left and 50,50 = center), "
|
||||
"read off the screenshot. Optional hold_ms presses and holds (e.g. a "
|
||||
"charge-up or a platformer jump). For element-based pages prefer "
|
||||
"BrowserClickIndex; use this when there is nothing in the element list "
|
||||
"to click."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"xPercent": {
|
||||
"type": "number",
|
||||
"description": "Horizontal position as a percent of view width (0=left, 100=right).",
|
||||
},
|
||||
"yPercent": {
|
||||
"type": "number",
|
||||
"description": "Vertical position as a percent of view height (0=top, 100=bottom).",
|
||||
},
|
||||
"hold_ms": {
|
||||
"type": "number",
|
||||
"description": "Optional. Milliseconds to hold the button down before releasing (default 0 = a tap). Max 5000.",
|
||||
},
|
||||
"button": {
|
||||
"type": "string",
|
||||
"enum": ["left", "right", "middle"],
|
||||
"description": "Mouse button; defaults to left.",
|
||||
},
|
||||
},
|
||||
"required": ["xPercent", "yPercent"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "BrowserWait",
|
||||
"description": (
|
||||
@@ -644,7 +683,7 @@ BROWSER_TOOLS_SCHEMA = [
|
||||
]
|
||||
|
||||
# Schema-forced batching: the model ignored every prompt-level batching invitation (0 adoptions across 8 measured runs), so the single-step mutating tools are not offered to it at all; acting means a BrowserBatch array, and the one deliberate solo path is BrowserClickIndex (irreversible step with expect, or a text-box fill). Executors and replay still support everything.
|
||||
P_SOLO_MUTATORS_HIDDEN = {"BrowserNavigate", "BrowserClick", "BrowserType", "BrowserScroll", "BrowserPressKey"}
|
||||
P_SOLO_MUTATORS_HIDDEN = {"BrowserNavigate", "BrowserClick", "BrowserType", "BrowserScroll", "BrowserPressKey", "BrowserClickPoint"}
|
||||
MODEL_VISIBLE_TOOLS = [t for t in BROWSER_TOOLS_SCHEMA if t["name"] not in P_SOLO_MUTATORS_HIDDEN]
|
||||
|
||||
ACTION_MAP = {
|
||||
@@ -661,6 +700,7 @@ ACTION_MAP = {
|
||||
"BrowserPressKey": "press_key",
|
||||
"BrowserListInteractives": "list_interactives",
|
||||
"BrowserClickIndex": "click_index",
|
||||
"BrowserClickPoint": "click_point",
|
||||
"BrowserBatch": "batch",
|
||||
"BrowserDetectWebMCP": "detect_webmcp",
|
||||
"BrowserListRoutes": "list_routes",
|
||||
@@ -669,6 +709,89 @@ ACTION_MAP = {
|
||||
"BrowserClickByName": "click_by_name",
|
||||
}
|
||||
|
||||
# --- App agent: driving an OpenSwarm-built app via its native bridge ---------
|
||||
# Apps expose window.OPENSWARM_APP = { describe(), getState(), invoke(name,args) }.
|
||||
# The agent reads structure/state and acts through that bridge in single
|
||||
# executeJavaScript round-trips, no screenshots or accessibility tree. These
|
||||
# three tools are translated to BrowserEvaluate in execute_browser_tool, so they
|
||||
# need no frontend command-handler changes.
|
||||
APP_TOOLS_SCHEMA = [
|
||||
{
|
||||
"name": "AppDescribe",
|
||||
"description": (
|
||||
"Read the app's rules and CURRENT list of actions, straight from the "
|
||||
"app itself (window.OPENSWARM_APP.describe()). Returns "
|
||||
"{rules, controls, __rev}: rules is what the app is and its objective, "
|
||||
"controls is an array of {name, args, description, keys}, and __rev is "
|
||||
"a revision number. (Older apps may return a bare array of controls.) "
|
||||
"These are ALREADY front-loaded into your first message, so you rarely "
|
||||
"need to call this. The app's controls are DYNAMIC: call this again "
|
||||
"ONLY when AppGetState reports a changed __rev (e.g. after an AppInvoke "
|
||||
"added or removed actions). Returns null if the app exposes no bridge "
|
||||
"(then fall back to BrowserListInteractives/BrowserScreenshot)."
|
||||
),
|
||||
"input_schema": {"type": "object", "properties": {}, "required": []},
|
||||
},
|
||||
{
|
||||
"name": "AppGetState",
|
||||
"description": (
|
||||
"Read a small JSON snapshot of the app's current state "
|
||||
"(window.OPENSWARM_APP.getState()). Use it to check what's on screen "
|
||||
"and to verify an action landed. The snapshot includes __rev, the "
|
||||
"controls revision: if it differs from the __rev you were given, the "
|
||||
"controls changed, so call AppDescribe to refresh them. Returns null "
|
||||
"if the bridge is absent."
|
||||
),
|
||||
"input_schema": {"type": "object", "properties": {}, "required": []},
|
||||
},
|
||||
{
|
||||
"name": "AppInvoke",
|
||||
"description": (
|
||||
"Perform one app action by name with arguments "
|
||||
"(window.OPENSWARM_APP.invoke(name, args)). The name MUST be one that "
|
||||
"AppDescribe just returned; never invent or modify actions, only invoke "
|
||||
"the ones the app exposes. Returns the action's result, or an "
|
||||
"{__error__} if it threw. After invoking, re-AppDescribe if the action "
|
||||
"may have changed which controls exist."
|
||||
),
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {"type": "string", "description": "Action name from AppDescribe."},
|
||||
"args": {
|
||||
"type": "object",
|
||||
"description": "Arguments object for the action; {} if none.",
|
||||
},
|
||||
},
|
||||
"required": ["name"],
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
# Bridge tool names -> the JS the app exposes. execute_browser_tool maps these to
|
||||
# a BrowserEvaluate. Each expression returns a JSON string (so it round-trips as
|
||||
# text) and never throws (errors are returned as JSON).
|
||||
APP_BRIDGE_TOOLS = {"AppDescribe", "AppGetState", "AppInvoke"}
|
||||
|
||||
# Lean toolset for app mode: bridge tools first, then a UI-driving fallback for
|
||||
# apps that don't expose the bridge. Built by name-selecting the shared defs so
|
||||
# their schemas stay in one place.
|
||||
P_APP_FALLBACK_TOOL_NAMES = [
|
||||
"ReportProgress", "Done",
|
||||
"BrowserScreenshot", "BrowserGetText",
|
||||
"BrowserListInteractives", "BrowserClickIndex", "BrowserBatch",
|
||||
# Human-style native input for apps with no clickable DOM (canvas games):
|
||||
# press real keys and tap real screen points the way a person plays.
|
||||
"BrowserPressKey", "BrowserClickPoint",
|
||||
]
|
||||
p_app_fallback_tools = [t for t in BROWSER_TOOLS_SCHEMA if t["name"] in P_APP_FALLBACK_TOOL_NAMES]
|
||||
# ReportProgress + Done lead, then the bridge tools, then UI fallback.
|
||||
APP_VISIBLE_TOOLS = (
|
||||
[t for t in p_app_fallback_tools if t["name"] in ("ReportProgress", "Done")]
|
||||
+ APP_TOOLS_SCHEMA
|
||||
+ [t for t in p_app_fallback_tools if t["name"] not in ("ReportProgress", "Done")]
|
||||
)
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You are a website-agnostic browser automation agent. You can operate on ANY "
|
||||
"website the user is signed into; social media, dating apps, email, productivity "
|
||||
@@ -895,6 +1018,96 @@ SYSTEM_PROMPT = (
|
||||
|
||||
MAX_TURNS = 40
|
||||
|
||||
# App mode: drive an OpenSwarm-built app through its native bridge. This is the
|
||||
# global "how to operate an app" guidance (decision: one global doc, not per-app;
|
||||
# per-app specifics come live from AppDescribe). Deliberately short, the bridge
|
||||
# does the heavy lifting and future models need less hand-holding.
|
||||
APP_SYSTEM_PROMPT = (
|
||||
"You operate an OpenSwarm-built app (a small web app the user created, e.g. a "
|
||||
"graphing tool or a form). The app is ALREADY open in front of you; do not "
|
||||
"navigate anywhere.\n\n"
|
||||
|
||||
"## How you see and act: the app's own bridge (this is the fast path)\n"
|
||||
"The app exposes a native bridge, window.OPENSWARM_APP, with three calls you "
|
||||
"reach through tools:\n"
|
||||
"- AppDescribe -> {rules, controls, __rev}: the app's objective and its "
|
||||
"current actions {name, args, description, keys}.\n"
|
||||
"- AppGetState -> a small JSON snapshot of what's on screen (includes __rev).\n"
|
||||
"- AppInvoke(name, args) -> perform one action.\n"
|
||||
"The app's rules and controls have ALREADY been read for you and placed in "
|
||||
"your first message, so you can start invoking actions immediately; you do "
|
||||
"NOT need screenshots, the DOM, or the accessibility tree, and you usually do "
|
||||
"NOT need to call AppDescribe at all.\n"
|
||||
"Only ever call actions that AppDescribe actually returned. You operate the app, "
|
||||
"you do NOT change it: never invent action names, and never try to add, remove, "
|
||||
"or redefine the app's available actions or edit its code. If what the user wants "
|
||||
"isn't reachable through the exposed actions, say so in Done.\n\n"
|
||||
|
||||
"## Controls are DYNAMIC, but you only re-read on a __rev change\n"
|
||||
"Actions can change as you interact (the app adds and removes controls). The "
|
||||
"front-loaded controls came with a __rev number. Use AppGetState to verify "
|
||||
"outcomes; if its __rev differs from the one you have, the controls changed, "
|
||||
"so call AppDescribe ONCE to refresh them. As long as __rev is unchanged, "
|
||||
"trust the controls you already have and do not re-describe.\n\n"
|
||||
|
||||
"## Fast/real-time games: turn on autopilot, then SUPERVISE\n"
|
||||
"If the controls include `__autopilot__`, the app can play itself at frame "
|
||||
"rate (twitch games like Flappy or Doodle Jump are impossible to play by "
|
||||
"pressing a key per screenshot; the network round-trip is far slower than the "
|
||||
"game). Do NOT try to react frame by frame. Instead run two loops:\n"
|
||||
"- Start it: AppInvoke('__autopilot__', {on:true}). The app's reflex now "
|
||||
"handles the per-frame twitch.\n"
|
||||
"- Confirm the reflex is ALIVE: AppGetState's __autopilotFrames must CLIMB "
|
||||
"between polls. If it stays flat while __autopilot is true, the frame loop is "
|
||||
"throttled (the app is not running at speed), not mis-tuned; report that "
|
||||
"plainly rather than fighting it with knobs.\n"
|
||||
"- Supervise on a SLOW cadence (every few seconds): AppGetState and check "
|
||||
"progressRate / alive / score. Healthy play = cheap state polls; do NOT "
|
||||
"screenshot every turn.\n"
|
||||
"- Escalate only when stuck: if progressRate stays ~0 while alive across a "
|
||||
"couple of polls, take ONE BrowserScreenshot to see what the one-frame "
|
||||
"heuristic cannot (a platform off to the side, a hazard), then STEER by "
|
||||
"invoking __autopilot__ again with this app's steering knobs (named in the "
|
||||
"__autopilot__ control description / the app's rules), e.g. {bias:-30} or "
|
||||
"{aggressiveness:1.5}. The running reflex obeys at frame rate; you never "
|
||||
"stop it.\n"
|
||||
"- Stop when the goal is met (or to hand back control): "
|
||||
"AppInvoke('__autopilot__', {on:false}).\n\n"
|
||||
|
||||
"## If there is no bridge: operate it like a person\n"
|
||||
"If AppDescribe (or AppGetState) returns null, this app exposes no bridge, so "
|
||||
"drive it directly the way a human would, by looking at the screen and using "
|
||||
"the keyboard and mouse:\n"
|
||||
"- SEE with BrowserScreenshot (your main sense here); read on-screen text/score "
|
||||
"from it. BrowserGetText helps for text-heavy apps.\n"
|
||||
"- For a normal HTML app (buttons, inputs, forms): BrowserListInteractives to "
|
||||
"find controls, then BrowserClickIndex / BrowserBatch to act.\n"
|
||||
"- For a CANVAS app or GAME (no clickable elements in the list): play it with "
|
||||
"native input. BrowserPressKey for keyboard (e.g. 'Space' to flap, "
|
||||
"'ArrowLeft'/'ArrowRight' to move, 'Enter' to start) and BrowserClickPoint to "
|
||||
"tap a spot, giving xPercent/yPercent read off the screenshot (50,50 = center). "
|
||||
"These are REAL OS-level events, identical to you pressing a key or clicking, so "
|
||||
"the game responds exactly as it does for a person. Take a screenshot to "
|
||||
"confirm what changed, then act again.\n"
|
||||
"- Need fast repeated input (rapid flaps/taps)? Put several BrowserPressKey or "
|
||||
"BrowserClickPoint steps in one BrowserBatch so they fire in a single turn.\n"
|
||||
"If you truly cannot operate it, say so plainly in Done with success=false.\n\n"
|
||||
|
||||
"## ReportProgress before acting\n"
|
||||
"Before any AppInvoke (or UI action), call ReportProgress in the SAME turn with "
|
||||
"a telegraphic next_goal and working_memory. AppDescribe and AppGetState are "
|
||||
"reads and do not require it.\n\n"
|
||||
|
||||
"## Speed\n"
|
||||
"Fewer model turns is the #1 driver. Once AppDescribe tells you the actions, "
|
||||
"fire the AppInvokes you need; don't re-describe between every step unless the "
|
||||
"action list could have changed. Keep your notes a few words each.\n\n"
|
||||
|
||||
"When finished, end by calling Done; put a plain one or two sentence reply in "
|
||||
"its message (what you did and the proof, e.g. what's now graphed), zero "
|
||||
"interface words. Set success=false if you couldn't finish."
|
||||
)
|
||||
|
||||
# Tools that count as "action tools"; calling any of these in a turn requires the model to also call ReportProgress in the same turn (after the first turn). Read-only tools and meta tools are exempt.
|
||||
ACTION_TOOLS_REQUIRING_REPORT = {
|
||||
"BrowserClick",
|
||||
@@ -904,5 +1117,7 @@ ACTION_TOOLS_REQUIRING_REPORT = {
|
||||
"BrowserScroll",
|
||||
"BrowserEvaluate",
|
||||
"BrowserClickIndex", # Phase 3
|
||||
"BrowserClickPoint", # app mode: tap a canvas/game at a screen point
|
||||
"BrowserBatch", # Phase 4
|
||||
"AppInvoke", # app mode: invoking an app action mutates state
|
||||
}
|
||||
|
||||
@@ -22,6 +22,8 @@ MODEL = os.environ.get("OPENSWARM_AGENT_MODEL", "sonnet")
|
||||
DASHBOARD_ID = os.environ.get("OPENSWARM_DASHBOARD_ID", "")
|
||||
PRE_SELECTED_BROWSER_IDS = os.environ.get("OPENSWARM_PRE_SELECTED_BROWSER_IDS", "")
|
||||
PARENT_SESSION_ID = os.environ.get("OPENSWARM_PARENT_SESSION_ID", "")
|
||||
# Apps the user selected on the dashboard; AppAgent may only target these (anti-hallucination).
|
||||
SELECTED_APP_IDS = [a.strip() for a in os.environ.get("OPENSWARM_SELECTED_APP_IDS", "").split(",") if a.strip()]
|
||||
|
||||
TOOLS = [
|
||||
{
|
||||
@@ -112,6 +114,40 @@ TOOLS = [
|
||||
"required": ["tasks"],
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "AppAgent",
|
||||
"description": (
|
||||
"Operate one of the user's OpenSwarm-built apps (a small web app they "
|
||||
"created, e.g. a graphing tool, a form, or a canvas game like Doom) that "
|
||||
"is open on the dashboard. A dedicated app agent performs the task: it "
|
||||
"drives the app through its native bridge when one is available (reading "
|
||||
"the app's own state and calling its controls), and otherwise falls back "
|
||||
"to native keyboard/mouse plus screenshots for canvas and game apps that "
|
||||
"expose no bridge. It returns a summary plus a final screenshot. Works for "
|
||||
"ANY app in the selected-app context, including games and canvas apps; use "
|
||||
"BrowserAgent only for websites, not these apps."
|
||||
),
|
||||
"inputSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"output_id": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"The id of the selected app to operate (from the selected-app "
|
||||
"context block)."
|
||||
),
|
||||
},
|
||||
"task": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"What to do in the app. Be specific (e.g. 'graph y=x^2 and "
|
||||
"y=sin(x)')."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["output_id", "task"],
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@@ -245,37 +281,54 @@ def format_batch_results(results: list[dict]) -> dict:
|
||||
return {"content": all_content}
|
||||
|
||||
|
||||
def p_text_error(message: str) -> dict:
|
||||
return {"content": [{"type": "text", "text": message}], "isError": True}
|
||||
|
||||
|
||||
def p_run_single_task(task_def: dict) -> dict:
|
||||
"""Dispatch one task to the backend and format its single result."""
|
||||
result = call_backend([task_def])
|
||||
if "error" in result:
|
||||
return p_text_error(f"Error: {result['error']}")
|
||||
results = result.get("results", [result])
|
||||
if results:
|
||||
return format_result(results[0])
|
||||
return p_text_error("No result returned.")
|
||||
|
||||
|
||||
def handle_tool_call(tool_name: str, arguments: dict) -> dict:
|
||||
if tool_name == "CreateBrowserAgent":
|
||||
task_def = {
|
||||
return p_run_single_task({
|
||||
"task": arguments.get("task", ""),
|
||||
"browser_id": "",
|
||||
"url": arguments.get("url", ""),
|
||||
}
|
||||
result = call_backend([task_def])
|
||||
if "error" in result:
|
||||
return {"content": [{"type": "text", "text": f"Error: {result['error']}"}], "isError": True}
|
||||
results = result.get("results", [result])
|
||||
if results:
|
||||
return format_result(results[0])
|
||||
return {"content": [{"type": "text", "text": "No result returned."}], "isError": True}
|
||||
})
|
||||
|
||||
elif tool_name == "BrowserAgent":
|
||||
browser_id = arguments.get("browser_id", "")
|
||||
if not browser_id:
|
||||
return {"content": [{"type": "text", "text": "Error: browser_id is required"}], "isError": True}
|
||||
task_def = {
|
||||
return p_text_error("Error: browser_id is required")
|
||||
return p_run_single_task({
|
||||
"task": arguments.get("task", ""),
|
||||
"browser_id": browser_id,
|
||||
"url": "",
|
||||
}
|
||||
result = call_backend([task_def])
|
||||
if "error" in result:
|
||||
return {"content": [{"type": "text", "text": f"Error: {result['error']}"}], "isError": True}
|
||||
results = result.get("results", [result])
|
||||
if results:
|
||||
return format_result(results[0])
|
||||
return {"content": [{"type": "text", "text": "No result returned."}], "isError": True}
|
||||
})
|
||||
|
||||
elif tool_name == "AppAgent":
|
||||
output_id = arguments.get("output_id", "")
|
||||
if not output_id:
|
||||
return p_text_error("Error: output_id is required")
|
||||
# Only drive apps the user actually selected (anti-hallucination), when we
|
||||
# know the selection. Empty list = unknown, so don't block.
|
||||
if SELECTED_APP_IDS and output_id not in SELECTED_APP_IDS:
|
||||
valid = ", ".join(SELECTED_APP_IDS) or "(none)"
|
||||
return p_text_error(f"Error: '{output_id}' is not a selected app. Selected apps: {valid}")
|
||||
return p_run_single_task({
|
||||
"task": arguments.get("task", ""),
|
||||
"browser_id": f"app:{output_id}",
|
||||
"url": "",
|
||||
"app_mode": True,
|
||||
})
|
||||
|
||||
elif tool_name == "BrowserAgents":
|
||||
tasks = arguments.get("tasks", [])
|
||||
|
||||
@@ -159,6 +159,7 @@ def build_selected_app_context(selected_app_output_ids: Optional[List[str]]) ->
|
||||
name = output.name or "Untitled App"
|
||||
lines = [
|
||||
f'- App: "{name}"',
|
||||
f" App id (for AppAgent): {output_id}",
|
||||
f" Workspace path: {path}",
|
||||
f" Entry point: {os.path.join(path, 'index.html')}",
|
||||
]
|
||||
@@ -174,10 +175,14 @@ def build_selected_app_context(selected_app_output_ids: Optional[List[str]]) ->
|
||||
return None
|
||||
return (
|
||||
"<selected_app_context>\n"
|
||||
"The user selected these App cards on the dashboard for you to edit. "
|
||||
"They are existing web apps; edit the files in place at the paths below "
|
||||
"and the dashboard preview live-reloads on save. Do not scaffold a new "
|
||||
"project or write files anywhere else.\n\n"
|
||||
"The user selected these App cards on the dashboard. You can do two things "
|
||||
"with them:\n"
|
||||
"- EDIT them: change the files in place at the paths below and the dashboard "
|
||||
"preview live-reloads on save. Do not scaffold a new project or write files "
|
||||
"elsewhere.\n"
|
||||
"- OPERATE them: to actually use a running app (e.g. 'graph y=x^2', 'fill in "
|
||||
"the form'), call AppAgent(output_id, task) with the App id below. A "
|
||||
"dedicated agent drives the live app through its own actions, no editing.\n\n"
|
||||
+ "\n\n".join(entries)
|
||||
+ "\n</selected_app_context>"
|
||||
)
|
||||
|
||||
@@ -20,10 +20,11 @@ def register_builtin_mcp_servers(
|
||||
session: AgentSession,
|
||||
builtin_perms: Dict[str, str],
|
||||
selected_browser_ids: Optional[List[str]],
|
||||
selected_app_output_ids: Optional[List[str]],
|
||||
) -> Tuple[List[str], List[str]]:
|
||||
import backend.apps.agents as p_agents_pkg
|
||||
agents_dir = os.path.dirname(p_agents_pkg.__file__)
|
||||
browser_delegation_tools = ["CreateBrowserAgent", "BrowserAgent", "BrowserAgents"]
|
||||
browser_delegation_tools = ["CreateBrowserAgent", "BrowserAgent", "BrowserAgents", "AppAgent"]
|
||||
browser_all_denied = all(
|
||||
builtin_perms.get(t, "always_allow") == "deny"
|
||||
for t in browser_delegation_tools
|
||||
@@ -36,6 +37,8 @@ def register_builtin_mcp_servers(
|
||||
backend_port = os.environ.get("OPENSWARM_PORT", "8324")
|
||||
# Only the card the user actually picked in select-mode gets claimed for the task, so the sub drives that one instead of opening its own duplicate. Passing EVERY dashboard card here (the old behavior) made the sub force-grab a random, usually-parked card and never navigate it, which broke the bulk of browser tasks.
|
||||
pre_selected_bids = [b for b in (selected_browser_ids or []) if b]
|
||||
# Apps the user selected this turn; the AppAgent tool may only target these (anti-hallucination gate in the MCP server, which reads this at startup).
|
||||
selected_app_ids = [a for a in (selected_app_output_ids or []) if a]
|
||||
auth_tok = get_auth_token()
|
||||
mcp_servers["openswarm-browser-agent"] = {
|
||||
"command": sys.executable,
|
||||
@@ -46,6 +49,7 @@ def register_builtin_mcp_servers(
|
||||
"OPENSWARM_AGENT_MODEL": session.model,
|
||||
"OPENSWARM_DASHBOARD_ID": session.dashboard_id or "",
|
||||
"OPENSWARM_PRE_SELECTED_BROWSER_IDS": ",".join(pre_selected_bids),
|
||||
"OPENSWARM_SELECTED_APP_IDS": ",".join(selected_app_ids),
|
||||
"OPENSWARM_PARENT_SESSION_ID": session.id,
|
||||
},
|
||||
"type": "stdio",
|
||||
|
||||
@@ -104,7 +104,7 @@ class RunOptions(AgentManagerProtocol):
|
||||
mcp_servers = await self.build_mcp_servers(session.allowed_tools, session.active_mcps)
|
||||
|
||||
browser_delegation_tools, invoke_agent_tools = register_builtin_mcp_servers(
|
||||
mcp_servers, session, builtin_perms, selected_browser_ids
|
||||
mcp_servers, session, builtin_perms, selected_browser_ids, selected_app_output_ids
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -56,6 +56,13 @@ fast.
|
||||
3. Leave `frontend/package.json`, `frontend/vite.config.ts`, `run.sh`,
|
||||
`.env`, `meta.json` alone — vite still needs them.
|
||||
4. Don't run `bash backend_init.sh` — lightweight mode has no backend.
|
||||
5. Agent control still works without the template: `window.OPENSWARM_APP` is
|
||||
injected by the app shell, so even here you can make the app agent-operable
|
||||
by calling `window.OPENSWARM_APP.register({ rules, controls, getState,
|
||||
invoke })` from your inline `<script>` (see the bridge section below). This is
|
||||
optional, the agent can also play any app via native keyboard/mouse, but for
|
||||
a game/canvas a registered `getState` (e.g. score, alive) makes it far more
|
||||
reliable. Don't wire up the bridge object yourself; it already exists.
|
||||
|
||||
The rest of this document covers **workspace mode**. If you picked
|
||||
lightweight, only the "Debugging" section (frontend console logs in the
|
||||
@@ -364,6 +371,79 @@ things those two can't do (it won't be there once published).
|
||||
|
||||
---
|
||||
|
||||
## Make the app agent-operable: the `OPENSWARM_APP` bridge
|
||||
|
||||
An agent can drive this app on the user's behalf (e.g. "graph y=x^2 on my
|
||||
Desmos app"). It does NOT do that by clicking pixels or scraping the DOM (slow,
|
||||
and an app's DOM is often a bare `<canvas>`). Instead it reads and acts through
|
||||
`window.OPENSWARM_APP`, a bridge the template already ships for you
|
||||
(`src/agentBridge.ts`, installed before your app mounts).
|
||||
|
||||
**You do not wire up the bridge; you `register()` into it.** Call
|
||||
`window.OPENSWARM_APP.register({ rules, controls, getState, invoke })` once your
|
||||
app's core object exists (e.g. in a mount `useEffect`). This is REQUIRED for
|
||||
every app: the runtime verifies it and the agent's first action fails loudly
|
||||
with `BRIDGE MISSING` if you forget. Pass:
|
||||
|
||||
- `rules` (string) - what the app is and its objective, in plain prose. This is
|
||||
what the agent reads to understand the app (e.g. "Flappy Bird. Keep the bird
|
||||
airborne through the pipe gaps; the game ends on a collision.").
|
||||
- `controls` - an array of `{ name, args?, description?, keys? }`, OR a function
|
||||
returning that array when controls are dynamic. `keys` is an optional
|
||||
human-style hint (e.g. `"Space = flap"`). Return only what's available now.
|
||||
- `getState()` - a **small** JSON snapshot used to verify an action landed. Keep
|
||||
it compact; this is the latency budget.
|
||||
- `invoke(name, args)` - perform the named action and return a result (or throw
|
||||
a string the agent will read).
|
||||
|
||||
Keep `args` shapes simple (strings, numbers, booleans, small objects). The agent
|
||||
only ever calls actions that `controls` listed; it never edits your code. When
|
||||
dynamic controls change, call `window.OPENSWARM_APP.refresh()` so the agent
|
||||
knows to re-read them (it bumps the `__rev` the agent watches).
|
||||
|
||||
```tsx
|
||||
// `calc` here is the app's own API (Desmos example); use whatever yours exposes.
|
||||
function registerAgentBridge(calc: any) {
|
||||
window.OPENSWARM_APP!.register({
|
||||
rules: 'A graphing calculator. Plot and remove expressions like y=x^2 or y=sin(x).',
|
||||
controls() {
|
||||
const controls = [
|
||||
{ name: 'addExpr', args: { latex: 'string' }, description: 'Add a graph expression, e.g. y=x^2' },
|
||||
{ name: 'clear', description: 'Remove all expressions' },
|
||||
];
|
||||
// Dynamic: only offer removeExpr when something is on the graph.
|
||||
if (calc.getExpressions().length > 0) {
|
||||
controls.push({ name: 'removeExpr', args: { id: 'string' }, description: 'Remove one expression by id' });
|
||||
}
|
||||
return controls;
|
||||
},
|
||||
getState() {
|
||||
return { expressions: calc.getExpressions().map((e: any) => ({ id: e.id, latex: e.latex })) };
|
||||
},
|
||||
invoke(name: string, args: any = {}) {
|
||||
if (name === 'addExpr') { const id = String(Date.now()); calc.setExpression({ id, latex: args.latex }); window.OPENSWARM_APP!.refresh(); return { id }; }
|
||||
if (name === 'removeExpr') { calc.removeExpression({ id: args.id }); window.OPENSWARM_APP!.refresh(); return { ok: true }; }
|
||||
if (name === 'clear') { calc.setBlank(); window.OPENSWARM_APP!.refresh(); return { ok: true }; }
|
||||
throw `Unknown action: ${name}`;
|
||||
},
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
### Real-time games need a high-level action, not per-frame controls
|
||||
|
||||
An agent acts in discrete tool calls separated by network + model latency
|
||||
(hundreds of ms to seconds). It physically CANNOT hit frame-timing, so exposing
|
||||
only `invoke('flap')` makes a reflex game like Flappy Bird understandable but
|
||||
unwinnable: by the time the agent decides to flap, the bird has already fallen.
|
||||
For anything real-time, expose a **high-level action** the app executes on its
|
||||
own tick loop, e.g. `invoke('autopilot', { on: true })` that runs the optimal
|
||||
input internally, or `invoke('setDifficulty', ...)`. Let the agent set intent;
|
||||
let the app handle the milliseconds. Non-real-time apps (tools, forms, a
|
||||
Spotify-style player) don't need this: their actions are already at agent cadence.
|
||||
|
||||
---
|
||||
|
||||
## Debugging — use `swarm_debug`, not `print()`
|
||||
|
||||
The backend has `swarm_debug` pre-installed. It's a colored frame-aware
|
||||
|
||||
@@ -0,0 +1,195 @@
|
||||
// window.OPENSWARM_APP - the agent bridge, shipped with the template so it
|
||||
// EXISTS from first paint, before any app-specific code runs (index.tsx imports
|
||||
// this first). An app makes itself agent-operable by calling
|
||||
// window.OPENSWARM_APP.register({ rules, controls, getState, invoke }) on mount;
|
||||
// it never has to wire up the plumbing, so it cannot forget it. Until the app
|
||||
// registers, describe()/getState() report { __ready: false } so the agent knows
|
||||
// the app is still booting (and waits) instead of declaring it bridge-less.
|
||||
|
||||
export type AgentControl = {
|
||||
name: string;
|
||||
args?: Record<string, unknown>;
|
||||
description?: string;
|
||||
keys?: string; // optional key hint, e.g. "Space = flap", "WASD to move"
|
||||
};
|
||||
|
||||
export type AgentRegistration = {
|
||||
rules?: string; // what the app is and its objective, plain prose
|
||||
controls: AgentControl[] | (() => AgentControl[]); // a function for dynamic controls
|
||||
getState?: () => unknown; // small JSON snapshot, used to verify an action landed
|
||||
invoke: (name: string, args?: Record<string, unknown>) => unknown;
|
||||
// Optional in-page self-play for fast-twitch games (Flappy, Doodle Jump) the
|
||||
// agent cannot react to frame-by-frame over the network. This is the INNER loop
|
||||
// of a two-loop design: the bridge runs policy() at frame rate (reflexes), and
|
||||
// the agent supervises on a slow cadence (seconds) via getState. Each frame the
|
||||
// bridge calls policy(hint) and, if it returns a control name, invokes it. The
|
||||
// `hint` is the agent's steering channel: knobs it set with the reserved
|
||||
// "__autopilot__" control (e.g. {bias, aggressiveness, target}) so a slow
|
||||
// supervisor can correct the fast reflex without owning the frame loop. The app
|
||||
// owns the heuristic (reads its own live state); the bridge owns loop + timing.
|
||||
policy?: (hint: Record<string, unknown>) => string | null | undefined | void;
|
||||
};
|
||||
|
||||
type Bridge = {
|
||||
__openswarm: true;
|
||||
__ready: boolean;
|
||||
__rev: number;
|
||||
register: (api: AgentRegistration) => void;
|
||||
refresh: () => void; // bump __rev after dynamic controls change so the agent re-reads
|
||||
describe: () => unknown;
|
||||
getState: () => unknown;
|
||||
invoke: (name: string, args?: Record<string, unknown>) => unknown;
|
||||
};
|
||||
|
||||
declare global {
|
||||
interface Window {
|
||||
OPENSWARM_APP?: Bridge;
|
||||
}
|
||||
}
|
||||
|
||||
let registration: AgentRegistration | null = null;
|
||||
|
||||
// Reserved control name the agent invokes to start/stop/steer in-page self-play.
|
||||
const AUTOPILOT = '__autopilot__';
|
||||
let autopilotRAF = 0;
|
||||
// Frames the current autopilot run has executed. Surfaced in getState as
|
||||
// __autopilotFrames so a supervisor (or a human) can tell, from ONE poll, whether
|
||||
// the reflex loop is actually ticking: climbing == running at frame rate; stuck at
|
||||
// a low number while __autopilot is true == requestAnimationFrame is throttled
|
||||
// (e.g. the app webview is backgrounded), which no policy tuning can fix.
|
||||
let autopilotFrames = 0;
|
||||
// The agent's steering knobs, read by policy() each frame. Merged from the
|
||||
// non-`on` args of __autopilot__ invokes; the supervisor adjusts these after it
|
||||
// diagnoses a stall, the reflex obeys at frame rate.
|
||||
let autopilotHint: Record<string, unknown> = {};
|
||||
|
||||
function resolveControls(): AgentControl[] {
|
||||
if (!registration) return [];
|
||||
const c = registration.controls;
|
||||
try {
|
||||
return typeof c === 'function' ? c() || [] : c || [];
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
function autopilotRunning(): boolean {
|
||||
return autopilotRAF !== 0;
|
||||
}
|
||||
|
||||
function startAutopilot(): void {
|
||||
if (autopilotRAF || !registration || typeof registration.policy !== 'function') return;
|
||||
autopilotFrames = 0; // fresh run starts the frame count from zero
|
||||
const step = () => {
|
||||
// Re-arm the next frame FIRST so a single throwing frame can't kill the loop.
|
||||
autopilotRAF = requestAnimationFrame(step);
|
||||
autopilotFrames++;
|
||||
try {
|
||||
const name = registration && registration.policy ? registration.policy(autopilotHint) : null;
|
||||
if (name) bridge.invoke(name);
|
||||
} catch {
|
||||
/* swallow: keep playing; the agent monitors progress via getState/score */
|
||||
}
|
||||
};
|
||||
autopilotRAF = requestAnimationFrame(step);
|
||||
}
|
||||
|
||||
function stopAutopilot(): void {
|
||||
if (autopilotRAF) {
|
||||
cancelAnimationFrame(autopilotRAF);
|
||||
autopilotRAF = 0;
|
||||
}
|
||||
}
|
||||
|
||||
const bridge: Bridge = {
|
||||
__openswarm: true,
|
||||
__ready: false,
|
||||
__rev: 0,
|
||||
register(api: AgentRegistration) {
|
||||
stopAutopilot(); // a fresh registration owns its own loop; drop any prior one
|
||||
autopilotHint = {};
|
||||
autopilotFrames = 0;
|
||||
registration = api;
|
||||
bridge.__ready = true;
|
||||
bridge.__rev += 1;
|
||||
},
|
||||
refresh() {
|
||||
bridge.__rev += 1;
|
||||
},
|
||||
describe() {
|
||||
if (!bridge.__ready || !registration) {
|
||||
return { __ready: false, __rev: bridge.__rev };
|
||||
}
|
||||
// Copy so advertising the autopilot control never mutates the app's array.
|
||||
const controls = [...resolveControls()];
|
||||
if (typeof registration.policy === 'function') {
|
||||
controls.push({
|
||||
name: AUTOPILOT,
|
||||
args: { on: true },
|
||||
description:
|
||||
'Self-play: the app plays itself at frame rate so you never press keys ' +
|
||||
"per frame. {on:true} starts, {on:false} stops. Pass this app's own " +
|
||||
'steering knobs (named in the app rules/state) to adjust the running ' +
|
||||
'policy without stopping it. Supervise on a slow cadence: poll getState; ' +
|
||||
'if progress stalls, take ONE screenshot to diagnose, then re-invoke ' +
|
||||
'with an adjusted knob.',
|
||||
});
|
||||
}
|
||||
return {
|
||||
rules: registration.rules || '',
|
||||
controls,
|
||||
__rev: bridge.__rev,
|
||||
};
|
||||
},
|
||||
getState() {
|
||||
if (!bridge.__ready || !registration) {
|
||||
return { __ready: false, __rev: bridge.__rev };
|
||||
}
|
||||
let state: unknown = {};
|
||||
try {
|
||||
state = registration.getState ? registration.getState() : {};
|
||||
} catch (e) {
|
||||
return { __error__: String((e as Error)?.message || e), __rev: bridge.__rev };
|
||||
}
|
||||
// Carry __rev alongside the app's own state so the agent can detect a
|
||||
// controls change with a single getState, without re-describing every turn.
|
||||
const out: Record<string, unknown> =
|
||||
state && typeof state === 'object' && !Array.isArray(state)
|
||||
? { ...(state as Record<string, unknown>) }
|
||||
: { value: state };
|
||||
// __autopilot/__hint let the slow supervisor see the reflex's on/off state
|
||||
// and the knobs in effect; only surfaced for apps that registered a policy.
|
||||
if (typeof registration.policy === 'function') {
|
||||
out.__autopilot = autopilotRunning();
|
||||
out.__autopilotFrames = autopilotFrames;
|
||||
out.__hint = autopilotHint;
|
||||
}
|
||||
out.__rev = bridge.__rev;
|
||||
return out;
|
||||
},
|
||||
invoke(name: string, args?: Record<string, unknown>) {
|
||||
if (!bridge.__ready || !registration) {
|
||||
throw 'OPENSWARM_APP not registered yet';
|
||||
}
|
||||
if (name === AUTOPILOT) {
|
||||
if (typeof registration.policy !== 'function') {
|
||||
return { error: 'this app registered no autopilot policy' };
|
||||
}
|
||||
const { on, ...knobs } = args || {};
|
||||
const hasKnobs = Object.keys(knobs).length > 0;
|
||||
// Merge steering knobs into the live hint the policy reads each frame.
|
||||
if (hasKnobs) autopilotHint = { ...autopilotHint, ...knobs };
|
||||
// Toggle: explicit `on` wins; a bare call (no knobs either) means "start".
|
||||
if (on !== undefined) {
|
||||
if (on) startAutopilot();
|
||||
else stopAutopilot();
|
||||
} else if (!hasKnobs) {
|
||||
startAutopilot();
|
||||
}
|
||||
return { autopilot: autopilotRunning(), hint: autopilotHint };
|
||||
}
|
||||
return registration.invoke(name, args || {});
|
||||
},
|
||||
};
|
||||
|
||||
window.OPENSWARM_APP = bridge;
|
||||
@@ -1,3 +1,7 @@
|
||||
// Install window.OPENSWARM_APP BEFORE anything else so the agent bridge exists
|
||||
// from first paint, even while React + the app are still mounting. The app fills
|
||||
// it in by calling window.OPENSWARM_APP.register(...) on mount.
|
||||
import './agentBridge';
|
||||
import React from 'react';
|
||||
import { createRoot } from 'react-dom/client';
|
||||
import Main from './app/Main';
|
||||
|
||||
@@ -73,6 +73,7 @@ WALK_SKIP_DIRS = frozenset({
|
||||
".pytest_cache",
|
||||
".mypy_cache",
|
||||
".ruff_cache",
|
||||
".openswarm",
|
||||
})
|
||||
|
||||
# Cap per-file response size at 256 KB. Hand-written source rarely exceeds this; auto-generated bundles routinely run into the MBs and they're not what the user/agent is editing. Anything over the cap returns a truncated stub the frontend treats as "open the file directly to see full contents."
|
||||
|
||||
@@ -40,6 +40,7 @@ BUILTIN_TOOLS: list[BuiltinTool] = [
|
||||
BuiltinTool(name="CreateBrowserAgent", description="Create a new browser and run a task on it", category="browser_delegation"),
|
||||
BuiltinTool(name="BrowserAgent", description="Delegate a browser task to an existing browser agent", category="browser_delegation"),
|
||||
BuiltinTool(name="BrowserAgents", description="Run multiple browser tasks in parallel on existing browsers", category="browser_delegation"),
|
||||
BuiltinTool(name="AppAgent", description="Operate one of the user's OpenSwarm-built apps through its native bridge or native input", category="browser_delegation"),
|
||||
# Browser action tools (Layer 2, what the sub-agent executes)
|
||||
BuiltinTool(name="BrowserScreenshot", description="Capture a screenshot of the browser page", category="browser_action"),
|
||||
BuiltinTool(name="BrowserNavigate", description="Navigate the browser to a URL", category="browser_action"),
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
// key-probe.js: inspect keyboard events INSIDE a game's guest webview.
|
||||
//
|
||||
// Why: the open question on the keyboard fix is whether tool-issued
|
||||
// sendInputEvent delivers correct key codes once the guest is focused, or
|
||||
// whether it arrives with an empty `code` / keyCode:0 (which games ignore) and
|
||||
// we must switch to CDP Input.dispatchKeyEvent. This probe answers it directly
|
||||
// instead of inferring from whether the sprite moved.
|
||||
//
|
||||
// Usage:
|
||||
// 1. Right-click the game webview -> Inspect -> Console (this is the GUEST
|
||||
// console, http://127.0.0.1:<port>/..., NOT the host app devtools).
|
||||
// 2. Paste this whole file and hit Enter. You'll see "[key-probe] armed".
|
||||
// 3. Fire one tool-issued press_key, e.g. { key: "ArrowRight", hold_ms: 800 }.
|
||||
// 4. Read the logged lines, or call __keyProbe.dump() for a table.
|
||||
//
|
||||
// What to look for per event:
|
||||
// - isTrusted: should be true (native OS-level event). false = a JS
|
||||
// dispatchEvent, not our native path.
|
||||
// - code: should be e.g. "ArrowRight". EMPTY string is the failure signal.
|
||||
// - keyCode/which: should be e.g. 39. 0 is the failure signal.
|
||||
// - holdMs (on keyup): wall-clock ms the key was held. ~0 means the hold
|
||||
// collapsed to an instant tap (down+up same tick); ~800 means hold worked.
|
||||
// Decision: clean code + nonzero keyCode -> stay on sendInputEvent. Empty
|
||||
// code / keyCode 0 -> move the keyboard path to CDP Input.dispatchKeyEvent.
|
||||
|
||||
(function () {
|
||||
if (window.__keyProbe) {
|
||||
try { window.__keyProbe.disarm(); } catch (_) {}
|
||||
}
|
||||
var events = [];
|
||||
var downAt = Object.create(null); // key -> timestamp, to measure hold duration
|
||||
|
||||
function row(e) {
|
||||
var now = (performance && performance.now) ? performance.now() : Date.now();
|
||||
var holdMs = null;
|
||||
if (e.type === 'keydown') {
|
||||
downAt[e.code || e.key] = now;
|
||||
} else if (e.type === 'keyup') {
|
||||
var k = e.code || e.key;
|
||||
if (downAt[k] != null) { holdMs = Math.round(now - downAt[k]); delete downAt[k]; }
|
||||
}
|
||||
var r = {
|
||||
type: e.type,
|
||||
key: e.key,
|
||||
code: e.code, // EMPTY = failure signal
|
||||
keyCode: e.keyCode, // 0 = failure signal
|
||||
which: e.which,
|
||||
isTrusted: e.isTrusted, // true = native; false = synthetic JS
|
||||
repeat: e.repeat,
|
||||
target: (e.target && (e.target.tagName || e.target.nodeName)) || '(none)',
|
||||
holdMs: holdMs, // only set on keyup
|
||||
};
|
||||
events.push(r);
|
||||
var warn = (r.code === '' || r.code == null) ? ' <-- EMPTY code'
|
||||
: (r.keyCode === 0) ? ' <-- keyCode 0' : '';
|
||||
console.log('[key-probe]', r.type, JSON.stringify(r) + warn);
|
||||
}
|
||||
|
||||
// Capture phase + on window so we see the event even if the game stops
|
||||
// propagation on its own listener.
|
||||
var types = ['keydown', 'keyup', 'keypress'];
|
||||
types.forEach(function (t) { window.addEventListener(t, row, true); });
|
||||
|
||||
window.__keyProbe = {
|
||||
events: events,
|
||||
dump: function () { try { console.table(events); } catch (_) { console.log(events); } return events; },
|
||||
clear: function () { events.length = 0; for (var k in downAt) delete downAt[k]; console.log('[key-probe] cleared'); },
|
||||
disarm: function () { types.forEach(function (t) { window.removeEventListener(t, row, true); }); console.log('[key-probe] disarmed'); },
|
||||
};
|
||||
|
||||
console.log('[key-probe] armed on', location.href, '- fire a press_key, then __keyProbe.dump(). document.activeElement =', document.activeElement && (document.activeElement.tagName || document.activeElement.nodeName));
|
||||
})();
|
||||
@@ -474,6 +474,14 @@ async def browser_agent_run(request: Request):
|
||||
if not tasks:
|
||||
return JSONResponse({"error": "tasks array is required"}, status_code=400)
|
||||
|
||||
# [app-agent] step trace: log app-mode dispatches arriving at the route.
|
||||
for p_t in tasks:
|
||||
if p_t.get("app_mode") or str(p_t.get("browser_id", "")).startswith("app:"):
|
||||
logger.info(
|
||||
f"[app-agent] ROUTE /run: browser_id={p_t.get('browser_id')!r} "
|
||||
f"app_mode={p_t.get('app_mode')} task={str(p_t.get('task',''))[:120]!r}"
|
||||
)
|
||||
|
||||
results = await run_browser_agents(
|
||||
tasks=tasks,
|
||||
model=model,
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
"""App agent: driving an OpenSwarm-built app through its window.OPENSWARM_APP bridge.
|
||||
|
||||
Pins the wiring that makes an app drivable without touching a real webview or LLM:
|
||||
- the three App bridge tools translate to a single BrowserEvaluate against the
|
||||
app's bridge,
|
||||
- run_browser_agents forwards app_mode + the "app:<id>" target without creating
|
||||
or navigating a browser card,
|
||||
- the AppAgent delegation tool builds the right task and gates on the user's
|
||||
selected apps,
|
||||
- the orchestrator's selected-app context advertises AppAgent.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
|
||||
from backend.apps.agents.browser import browser_agent as BA
|
||||
|
||||
|
||||
# --- bridge expression -------------------------------------------------------
|
||||
def test_app_bridge_expression_describe_and_state():
|
||||
desc = BA.app_bridge_expression("AppDescribe", {})
|
||||
assert "window.OPENSWARM_APP.describe()" in desc
|
||||
assert "JSON.stringify" in desc # round-trips as text
|
||||
state = BA.app_bridge_expression("AppGetState", {})
|
||||
assert "window.OPENSWARM_APP.getState()" in state
|
||||
|
||||
|
||||
def test_app_bridge_expression_invoke_serializes_args():
|
||||
expr = BA.app_bridge_expression("AppInvoke", {"name": "addExpr", "args": {"latex": "y=x^2"}})
|
||||
# name + args are JSON-encoded into the call
|
||||
assert 'window.OPENSWARM_APP.invoke("addExpr", {"latex": "y=x^2"})' in expr
|
||||
# missing bridge is handled inside the expression, never throws
|
||||
assert "typeof A.describe!=='function'" in expr
|
||||
|
||||
|
||||
def test_app_bridge_expression_invoke_defaults_args_to_empty_object():
|
||||
expr = BA.app_bridge_expression("AppInvoke", {"name": "clear"})
|
||||
assert 'window.OPENSWARM_APP.invoke("clear", {})' in expr
|
||||
|
||||
|
||||
# --- execute_browser_tool routing -------------------------------------------
|
||||
def test_execute_browser_tool_app_bridge_routes_to_evaluate(monkeypatch):
|
||||
captured = {}
|
||||
|
||||
async def p_send(request_id, action, browser_id, params, tab_id=""):
|
||||
captured.update(action=action, browser_id=browser_id, params=params)
|
||||
return {"text": json.dumps([{"name": "addExpr"}])}
|
||||
|
||||
monkeypatch.setattr(BA.ws_manager, "send_browser_command", p_send, raising=False)
|
||||
out = asyncio.run(BA.execute_browser_tool(
|
||||
"AppInvoke", {"name": "addExpr", "args": {"latex": "y=x^2"}}, "app:abc",
|
||||
))
|
||||
assert captured["action"] == "evaluate"
|
||||
assert captured["browser_id"] == "app:abc"
|
||||
assert "window.OPENSWARM_APP.invoke" in captured["params"]["expression"]
|
||||
assert out["text"] # result passes straight through
|
||||
|
||||
|
||||
# --- run_browser_agents app_mode wiring -------------------------------------
|
||||
def test_run_browser_agents_app_mode_forwards_flag_no_card(monkeypatch):
|
||||
recorded = {}
|
||||
|
||||
async def p_fake_run_browser_agent(**kwargs):
|
||||
recorded.update(kwargs)
|
||||
return {"summary": "done", "action_log": [], "final_screenshot": None}
|
||||
|
||||
async def p_boom(*a, **k):
|
||||
raise AssertionError("app mode must not create a browser card")
|
||||
|
||||
monkeypatch.setattr(BA, "run_browser_agent", p_fake_run_browser_agent, raising=True)
|
||||
monkeypatch.setattr(BA, "p_create_browser_card", p_boom, raising=True)
|
||||
# a connected dashboard so dispatch isn't refused
|
||||
monkeypatch.setattr(BA.ws_manager, "global_connections", [object()], raising=False)
|
||||
|
||||
results = asyncio.run(BA.run_browser_agents(
|
||||
tasks=[{"task": "graph y=x^2", "browser_id": "app:abc", "app_mode": True}],
|
||||
model="sonnet",
|
||||
dashboard_id="dash-1",
|
||||
))
|
||||
|
||||
assert results and results[0]["summary"] == "done"
|
||||
assert recorded["app_mode"] is True
|
||||
assert recorded["browser_id"] == "app:abc"
|
||||
assert recorded["initial_url"] is None # app already loaded; never navigate
|
||||
|
||||
|
||||
# --- AppAgent delegation tool -----------------------------------------------
|
||||
def p_load_mcp_server(monkeypatch, selected):
|
||||
import backend.apps.agents.browser_agent_mcp_server as srv
|
||||
monkeypatch.setattr(srv, "SELECTED_APP_IDS", list(selected), raising=False)
|
||||
return srv
|
||||
|
||||
|
||||
def test_app_agent_tool_builds_app_task(monkeypatch):
|
||||
srv = p_load_mcp_server(monkeypatch, ["abc"])
|
||||
captured = {}
|
||||
|
||||
def p_call_backend(tasks):
|
||||
captured["tasks"] = tasks
|
||||
return {"results": [{"summary": "graphed it", "action_log": []}]}
|
||||
|
||||
monkeypatch.setattr(srv, "call_backend", p_call_backend, raising=True)
|
||||
res = srv.handle_tool_call("AppAgent", {"output_id": "abc", "task": "graph y=x^2"})
|
||||
|
||||
assert not res.get("isError")
|
||||
task_def = captured["tasks"][0]
|
||||
assert task_def["browser_id"] == "app:abc"
|
||||
assert task_def["app_mode"] is True
|
||||
assert task_def["task"] == "graph y=x^2"
|
||||
|
||||
|
||||
def test_app_agent_tool_rejects_unselected_app(monkeypatch):
|
||||
srv = p_load_mcp_server(monkeypatch, ["abc"])
|
||||
|
||||
def p_call_backend(tasks):
|
||||
raise AssertionError("must not dispatch an unselected app")
|
||||
|
||||
monkeypatch.setattr(srv, "call_backend", p_call_backend, raising=True)
|
||||
res = srv.handle_tool_call("AppAgent", {"output_id": "zzz", "task": "graph"})
|
||||
assert res.get("isError")
|
||||
assert "not a selected app" in res["content"][0]["text"]
|
||||
|
||||
|
||||
def test_app_agent_tool_requires_output_id(monkeypatch):
|
||||
srv = p_load_mcp_server(monkeypatch, [])
|
||||
res = srv.handle_tool_call("AppAgent", {"task": "graph"})
|
||||
assert res.get("isError")
|
||||
assert "output_id is required" in res["content"][0]["text"]
|
||||
|
||||
|
||||
# --- orchestrator context advertises AppAgent --------------------------------
|
||||
def test_selected_app_context_advertises_app_agent(monkeypatch):
|
||||
import os
|
||||
from backend.apps.agents.manager.prompt import prompt_context as pc
|
||||
import backend.apps.outputs.workspace_io as wio
|
||||
|
||||
class POut:
|
||||
workspace_id = "ws-1"
|
||||
name = "Grapher"
|
||||
|
||||
monkeypatch.setattr(wio, "load_output", lambda oid: POut(), raising=True)
|
||||
monkeypatch.setattr(os.path, "isdir", lambda p: True, raising=True)
|
||||
monkeypatch.setattr(os.path, "isfile", lambda p: False, raising=True)
|
||||
|
||||
ctx = pc.build_selected_app_context(["abc"])
|
||||
assert ctx is not None
|
||||
assert "App id (for AppAgent): abc" in ctx
|
||||
assert "AppAgent(output_id, task)" in ctx
|
||||
|
||||
|
||||
# --- bridge readiness + parsing ---------------------------------------------
|
||||
def test_parse_bridge_result_decodes_json_text():
|
||||
assert BA.parse_bridge_result({"text": json.dumps([{"name": "x"}])}) == [{"name": "x"}]
|
||||
assert BA.parse_bridge_result({"text": "null"}) is None
|
||||
assert BA.parse_bridge_result({"error": "boom"}) is None
|
||||
assert BA.parse_bridge_result({"text": "not json"}) is None
|
||||
assert BA.parse_bridge_result({}) is None
|
||||
|
||||
|
||||
def test_bridge_ready_distinguishes_stub_from_registered():
|
||||
assert BA.bridge_ready([{"name": "x"}]) is True # legacy array
|
||||
assert BA.bridge_ready({"controls": [], "__rev": 1}) is True
|
||||
assert BA.bridge_ready({"__ready": False, "__rev": 0}) is False # template stub
|
||||
assert BA.bridge_ready({"__error__": "threw"}) is False
|
||||
assert BA.bridge_ready(None) is False
|
||||
|
||||
|
||||
def test_render_app_controls_array_and_object_forms():
|
||||
# Legacy array form: no rules, controls rendered.
|
||||
rules_md, controls_md = BA.render_app_controls([{"name": "clear", "description": "wipe"}])
|
||||
assert rules_md == ""
|
||||
assert "- `clear`: wipe" in controls_md
|
||||
|
||||
# New object form: rules + keys + args rendered.
|
||||
rules_md, controls_md = BA.render_app_controls({
|
||||
"rules": "Flappy Bird. Keep the bird airborne.",
|
||||
"controls": [{"name": "flap", "keys": "Space = flap", "args": {"force": "number"}, "description": "Flap once"}],
|
||||
"__rev": 3,
|
||||
})
|
||||
assert "Flappy Bird" in rules_md
|
||||
assert "- `flap`" in controls_md
|
||||
assert "[Space = flap]" in controls_md
|
||||
assert '"force"' in controls_md
|
||||
|
||||
# Not-ready / absent bridge yields nothing to render.
|
||||
assert BA.render_app_controls({"__ready": False}) is None
|
||||
assert BA.render_app_controls(None) is None
|
||||
|
||||
|
||||
# --- AppDescribe waits for a still-booting bridge ----------------------------
|
||||
def test_app_describe_polls_until_bridge_ready(monkeypatch):
|
||||
calls = {"n": 0}
|
||||
ready = {"rules": "r", "controls": [{"name": "x"}], "__rev": 1}
|
||||
|
||||
async def p_send(request_id, action, browser_id, params, tab_id=""):
|
||||
calls["n"] += 1
|
||||
if calls["n"] < 3:
|
||||
return {"text": json.dumps({"__ready": False, "__rev": 0})}
|
||||
return {"text": json.dumps(ready)}
|
||||
|
||||
async def p_no_sleep(p_s):
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(BA.ws_manager, "send_browser_command", p_send, raising=False)
|
||||
monkeypatch.setattr(BA.asyncio, "sleep", p_no_sleep, raising=True)
|
||||
monkeypatch.setattr(BA, "p_persist_app_controls", lambda *a, **k: None, raising=True)
|
||||
|
||||
out = asyncio.run(BA.execute_browser_tool("AppDescribe", {}, "app:abc"))
|
||||
assert calls["n"] == 3 # polled twice, succeeded on the third
|
||||
assert BA.parse_bridge_result(out) == ready
|
||||
|
||||
|
||||
def test_app_invoke_does_not_poll(monkeypatch):
|
||||
calls = {"n": 0}
|
||||
|
||||
async def p_send(request_id, action, browser_id, params, tab_id=""):
|
||||
calls["n"] += 1
|
||||
return {"text": json.dumps({"__ready": False})} # would loop forever if AppInvoke waited
|
||||
|
||||
async def p_no_sleep(p_s):
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(BA.ws_manager, "send_browser_command", p_send, raising=False)
|
||||
monkeypatch.setattr(BA.asyncio, "sleep", p_no_sleep, raising=True)
|
||||
|
||||
asyncio.run(BA.execute_browser_tool("AppInvoke", {"name": "flap"}, "app:abc"))
|
||||
assert calls["n"] == 1 # single shot, no readiness wait
|
||||
|
||||
|
||||
def test_app_describe_persists_controls_cache(monkeypatch, tmp_path):
|
||||
ready = {"rules": "Keep the bird airborne.", "controls": [{"name": "flap", "keys": "Space"}], "__rev": 1}
|
||||
|
||||
async def p_send(request_id, action, browser_id, params, tab_id=""):
|
||||
return {"text": json.dumps(ready)}
|
||||
|
||||
monkeypatch.setattr(BA.ws_manager, "send_browser_command", p_send, raising=False)
|
||||
monkeypatch.setattr(BA, "p_app_workspace_dir", lambda bid: str(tmp_path), raising=True)
|
||||
|
||||
asyncio.run(BA.execute_browser_tool("AppDescribe", {}, "app:abc"))
|
||||
controls = (tmp_path / ".openswarm" / "controls.md").read_text()
|
||||
rules = (tmp_path / ".openswarm" / "rules.md").read_text()
|
||||
assert "- `flap`" in controls and "[Space]" in controls
|
||||
assert "Keep the bird airborne." in rules
|
||||
@@ -15,14 +15,14 @@ def p_session():
|
||||
def test_registers_always_on_and_delegation_servers():
|
||||
mcp_servers = {}
|
||||
browser_tools, invoke_tools = register_builtin_mcp_servers(
|
||||
mcp_servers, p_session(), {}, None)
|
||||
mcp_servers, p_session(), {}, None, None)
|
||||
# always-on
|
||||
assert "openswarm-mcp-meta" in mcp_servers
|
||||
assert "openswarm-settings-meta" in mcp_servers
|
||||
# delegation (not denied)
|
||||
assert "openswarm-browser-agent" in mcp_servers
|
||||
assert "openswarm-invoke-agent" in mcp_servers
|
||||
assert browser_tools == ["CreateBrowserAgent", "BrowserAgent", "BrowserAgents"]
|
||||
assert browser_tools == ["CreateBrowserAgent", "BrowserAgent", "BrowserAgents", "AppAgent"]
|
||||
assert invoke_tools == ["InvokeAgent"]
|
||||
# Every registered server's script path must resolve to a file that ACTUALLY EXISTS. This is the assertion that catches a moved-caller resolving the wrong agents dir.
|
||||
for name in ("openswarm-mcp-meta", "openswarm-settings-meta",
|
||||
@@ -33,8 +33,8 @@ def test_registers_always_on_and_delegation_servers():
|
||||
|
||||
def test_fully_denied_delegation_servers_are_not_registered():
|
||||
mcp_servers = {}
|
||||
perms = {t: "deny" for t in ("CreateBrowserAgent", "BrowserAgent", "BrowserAgents", "InvokeAgent")}
|
||||
register_builtin_mcp_servers(mcp_servers, p_session(), perms, None)
|
||||
perms = {t: "deny" for t in ("CreateBrowserAgent", "BrowserAgent", "BrowserAgents", "AppAgent", "InvokeAgent")}
|
||||
register_builtin_mcp_servers(mcp_servers, p_session(), perms, None, None)
|
||||
assert "openswarm-browser-agent" not in mcp_servers # all browser tools denied -> skip
|
||||
assert "openswarm-invoke-agent" not in mcp_servers
|
||||
assert "openswarm-mcp-meta" in mcp_servers # always-on regardless
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
#!/bin/bash
|
||||
# watch-moves.sh: human-readable, moves-only view of the App Agent.
|
||||
# Shows just the actions it takes (clicks, keypresses, waits), one per line,
|
||||
# plus task-start and bridge-missing milestones. Hides all the dispatch/result/
|
||||
# electron echo noise. Usage: bash backend/watch-moves.sh [logfile]
|
||||
LOG="${1:-/tmp/openswarm.log}"
|
||||
tail -f "$LOG" | awk '
|
||||
{ gsub(/\033\[[0-9;]*m/, "") } # strip color codes
|
||||
match($0, /[0-9][0-9]:[0-9][0-9]:[0-9][0-9]/) { t = substr($0, RSTART, 8) }
|
||||
/\[app-agent\] START loop/ { print ""; print "=== " t " TASK START ==="; next }
|
||||
/BRIDGE MISSING/ { print t " !! bridge missing: app is NOT agent-operable, driving UI blind"; next }
|
||||
/\[browser-action\]/ {
|
||||
sub(/.*\[browser-action\] [A-Za-z]+: /, "") # drop everything up to the action
|
||||
sub(/ *-> .*/, "") # drop the trailing browser_id
|
||||
print t " > " $0
|
||||
next
|
||||
}
|
||||
'
|
||||
+89
-2
@@ -1719,12 +1719,18 @@ app.whenReady().then(async () => {
|
||||
try { app.dock.setIcon(iconPath); } catch (_) {}
|
||||
}
|
||||
|
||||
// NOTE: 'fullscreen' is deliberately NOT granted. A preview <webview>'s
|
||||
// content (App Builder apps) requesting HTML5 fullscreen gets promoted by
|
||||
// Electron into setFullScreen(true) on the top-level BrowserWindow, hijacking
|
||||
// the whole OpenSwarm window into true OS fullscreen on load. Denying the
|
||||
// permission breaks that chain; the user's own green-button fullscreen does
|
||||
// not route through here and is unaffected.
|
||||
session.defaultSession.setPermissionRequestHandler((_wc, permission, callback) => {
|
||||
const allowed = [
|
||||
'media', 'mediaKeySystem', 'protected-media-identifier',
|
||||
'geolocation', 'notifications', 'midi', 'midiSysex',
|
||||
'clipboard-read', 'clipboard-sanitized-write',
|
||||
'pointerLock', 'fullscreen', 'idle-detection',
|
||||
'pointerLock', 'idle-detection',
|
||||
];
|
||||
console.log('Permission request:', permission, '->', allowed.includes(permission) ? 'granted' : 'denied');
|
||||
callback(allowed.includes(permission));
|
||||
@@ -1733,7 +1739,7 @@ app.whenReady().then(async () => {
|
||||
const allowed = [
|
||||
'media', 'mediaKeySystem', 'protected-media-identifier',
|
||||
'clipboard-read', 'clipboard-sanitized-write',
|
||||
'pointerLock', 'fullscreen', 'idle-detection',
|
||||
'pointerLock', 'idle-detection',
|
||||
];
|
||||
return allowed.includes(permission);
|
||||
});
|
||||
@@ -2196,6 +2202,87 @@ app.on('web-contents-created', (_event, contents) => {
|
||||
})();
|
||||
`).catch(() => {});
|
||||
|
||||
// Agent bridge (window.OPENSWARM_APP). Injected into EVERY app's main world
|
||||
// from the shell so it exists regardless of frontend/src — the lightweight
|
||||
// App Builder mode deletes frontend/src (and with it the template's own
|
||||
// agentBridge.ts), so this is the only entry point a trimmed app can't lose.
|
||||
// Idempotent + guarded: a workspace app that imports its own bridge installs
|
||||
// first and this no-ops, so we never clobber a registered bridge. An app
|
||||
// becomes agent-operable by calling OPENSWARM_APP.register({rules, controls,
|
||||
// getState, invoke}); until it does, describe()/getState() report __ready:false
|
||||
// (the agent then falls back to native keyboard/mouse). Keep this in sync with
|
||||
// backend/apps/outputs/webapp_template/frontend/src/agentBridge.ts.
|
||||
contents.executeJavaScript(`
|
||||
(function() {
|
||||
if (window.OPENSWARM_APP) return;
|
||||
var registration = null;
|
||||
var AUTOPILOT = '__autopilot__';
|
||||
var autopilotRAF = 0;
|
||||
var autopilotFrames = 0;
|
||||
var autopilotHint = {};
|
||||
function resolveControls() {
|
||||
if (!registration) return [];
|
||||
var c = registration.controls;
|
||||
try { return (typeof c === 'function' ? c() : c) || []; } catch (e) { return []; }
|
||||
}
|
||||
function autopilotRunning() { return autopilotRAF !== 0; }
|
||||
function startAutopilot() {
|
||||
if (autopilotRAF || !registration || typeof registration.policy !== 'function') return;
|
||||
autopilotFrames = 0;
|
||||
var step = function() {
|
||||
autopilotRAF = requestAnimationFrame(step);
|
||||
autopilotFrames++;
|
||||
try {
|
||||
var name = registration && registration.policy ? registration.policy(autopilotHint) : null;
|
||||
if (name) bridge.invoke(name);
|
||||
} catch (e) {}
|
||||
};
|
||||
autopilotRAF = requestAnimationFrame(step);
|
||||
}
|
||||
function stopAutopilot() {
|
||||
if (autopilotRAF) { cancelAnimationFrame(autopilotRAF); autopilotRAF = 0; }
|
||||
}
|
||||
var bridge = {
|
||||
__openswarm: true, __ready: false, __rev: 0,
|
||||
register: function(api) { stopAutopilot(); autopilotHint = {}; autopilotFrames = 0; registration = api; bridge.__ready = true; bridge.__rev += 1; },
|
||||
refresh: function() { bridge.__rev += 1; },
|
||||
describe: function() {
|
||||
if (!bridge.__ready || !registration) return { __ready: false, __rev: bridge.__rev };
|
||||
var controls = resolveControls().slice();
|
||||
if (typeof registration.policy === 'function') {
|
||||
controls.push({ name: AUTOPILOT, args: { on: true }, description: "Self-play: the app plays itself at frame rate so you never press keys per frame. {on:true} starts, {on:false} stops. Pass this app's own steering knobs (named in the app rules/state) to adjust the running policy without stopping it. Supervise on a slow cadence: poll getState; if progress stalls, take ONE screenshot to diagnose, then re-invoke with an adjusted knob." });
|
||||
}
|
||||
return { rules: registration.rules || '', controls: controls, __rev: bridge.__rev };
|
||||
},
|
||||
getState: function() {
|
||||
if (!bridge.__ready || !registration) return { __ready: false, __rev: bridge.__rev };
|
||||
var state = {};
|
||||
try { state = registration.getState ? registration.getState() : {}; }
|
||||
catch (e) { return { __error__: String((e && e.message) || e), __rev: bridge.__rev }; }
|
||||
var out = (state && typeof state === 'object' && !Array.isArray(state)) ? Object.assign({}, state) : { value: state };
|
||||
if (typeof registration.policy === 'function') { out.__autopilot = autopilotRunning(); out.__autopilotFrames = autopilotFrames; out.__hint = autopilotHint; }
|
||||
out.__rev = bridge.__rev;
|
||||
return out;
|
||||
},
|
||||
invoke: function(name, args) {
|
||||
if (!bridge.__ready || !registration) throw 'OPENSWARM_APP not registered yet';
|
||||
if (name === AUTOPILOT) {
|
||||
if (typeof registration.policy !== 'function') return { error: 'this app registered no autopilot policy' };
|
||||
args = args || {};
|
||||
var on = args.on, hasKnobs = false;
|
||||
for (var k in args) { if (k !== 'on' && Object.prototype.hasOwnProperty.call(args, k)) { autopilotHint[k] = args[k]; hasKnobs = true; } }
|
||||
if (on !== undefined) { if (on) startAutopilot(); else stopAutopilot(); }
|
||||
else if (!hasKnobs) { startAutopilot(); }
|
||||
return { autopilot: autopilotRunning(), hint: autopilotHint };
|
||||
}
|
||||
return registration.invoke(name, args || {});
|
||||
},
|
||||
};
|
||||
window.OPENSWARM_APP = bridge;
|
||||
try { console.warn('[openswarm:bridge] shell-injected window.OPENSWARM_APP at', location.href); } catch (_) {}
|
||||
})();
|
||||
`).catch(() => {});
|
||||
|
||||
const url = contents.getURL();
|
||||
if (url.includes('spotify')) {
|
||||
contents.executeJavaScript(`
|
||||
|
||||
@@ -720,6 +720,7 @@ const DashboardOutputPreview: React.FC<{
|
||||
onConsoleMessage={handleConsoleMessage}
|
||||
interactive={interactive}
|
||||
onAppClicked={onAppClicked}
|
||||
agentBrowserId={`app:${output.id}`}
|
||||
/>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -6,6 +6,7 @@ import { useElementSelection } from '@/app/components/editor/ElementSelectionCon
|
||||
import { useIframeElementSelector } from './useIframeElementSelector';
|
||||
import { getAuthToken, ensureAuthToken } from '@/shared/config';
|
||||
import { useClaudeTokens } from '@/shared/styles/ThemeContext';
|
||||
import { registerWebview, unregisterWebview, setActiveTab, type BrowserWebview } from '@/shared/browserRegistry';
|
||||
|
||||
// In Electron use <webview> to escape iframe restrictions (popups, mic/camera, WebAuthn, cookied fetch); outside Electron fall back to iframe.
|
||||
const isElectron = navigator.userAgent.includes('Electron');
|
||||
@@ -57,8 +58,13 @@ interface Props {
|
||||
interactive?: boolean;
|
||||
/** Fired when the preload reports a mousedown inside the guest, so the host can flip the card into interactive mode. */
|
||||
onAppClicked?: () => void;
|
||||
/** When set (webview mode only), registers this preview's webview in browserRegistry under this id so an app agent can drive it via the browser command pipeline. */
|
||||
agentBrowserId?: string;
|
||||
}
|
||||
|
||||
// Single tab per app preview; the browser command pipeline keys on browserId:tabId.
|
||||
const APP_TAB_ID = 'main';
|
||||
|
||||
function buildSrcdoc(
|
||||
frontendCode: string,
|
||||
inputData: Record<string, any>,
|
||||
@@ -96,6 +102,7 @@ const ViewPreview = forwardRef<ViewPreviewHandle, Props>(({
|
||||
onContentLoad,
|
||||
interactive = false,
|
||||
onAppClicked,
|
||||
agentBrowserId,
|
||||
}, ref) => {
|
||||
const iframeRef = useRef<HTMLIFrameElement>(null);
|
||||
const webviewRef = useRef<any>(null);
|
||||
@@ -331,6 +338,23 @@ const ViewPreview = forwardRef<ViewPreviewHandle, Props>(({
|
||||
};
|
||||
}, [useWebview, handleNavigationLoad]);
|
||||
|
||||
// Register the live webview so an app agent can drive it through the same
|
||||
// browser command pipeline that drives browser cards. Webview-only: the iframe
|
||||
// fallback has no executeJavaScript channel from the host.
|
||||
useEffect(() => {
|
||||
if (!useWebview || !agentBrowserId) return;
|
||||
const wv = webviewRef.current as BrowserWebview | null;
|
||||
if (!wv) return;
|
||||
registerWebview(agentBrowserId, APP_TAB_ID, wv);
|
||||
setActiveTab(agentBrowserId, APP_TAB_ID);
|
||||
return () => {
|
||||
try { unregisterWebview(agentBrowserId, APP_TAB_ID); } catch (_e) {}
|
||||
};
|
||||
// Keyed on useWebview (flips true when the webview mounts), not iframeSrc:
|
||||
// the webview element has a stable key and survives src/data changes, so
|
||||
// re-registering on every data update would just thrash the registry.
|
||||
}, [useWebview, agentBrowserId]);
|
||||
|
||||
const hasContent = !!(serveUrl || frontendCode?.trim());
|
||||
|
||||
if (!hasContent) {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { getWebview, type BrowserWebview } from './browserRegistry';
|
||||
import { getWebview, registeredKeys, type BrowserWebview } from './browserRegistry';
|
||||
import { store } from './state/store';
|
||||
import { resumeBrowserCard } from './state/dashboardLayoutSlice';
|
||||
import { dashboardWs } from './ws/WebSocketManager';
|
||||
@@ -8,7 +8,7 @@ import { shouldStopWaiting, SETTLE_POLL_MS, settleProbeJs } from './browserSettl
|
||||
|
||||
let initialized = false;
|
||||
|
||||
export type BrowserAction = 'screenshot' | 'get_text' | 'get_console' | 'navigate' | 'click' | 'type' | 'evaluate' | 'get_elements' | 'scroll' | 'wait' | 'press_key' | 'list_interactives' | 'click_index' | 'batch' | 'detect_webmcp' | 'list_routes' | 'replay_route' | 'click_by_name';
|
||||
export type BrowserAction = 'screenshot' | 'get_text' | 'get_console' | 'navigate' | 'click' | 'type' | 'evaluate' | 'get_elements' | 'scroll' | 'wait' | 'press_key' | 'list_interactives' | 'click_index' | 'click_point' | 'batch' | 'detect_webmcp' | 'list_routes' | 'replay_route' | 'click_by_name';
|
||||
|
||||
export interface BrowserActivity {
|
||||
action: BrowserAction;
|
||||
@@ -63,6 +63,7 @@ const ACTION_LABELS: Record<string, string> = {
|
||||
press_key: 'Pressing key...',
|
||||
list_interactives: 'Reading page structure...',
|
||||
click_index: 'Clicking element...',
|
||||
click_point: 'Tapping screen...',
|
||||
click_by_name: 'Clicking element...',
|
||||
batch: 'Running batch...',
|
||||
};
|
||||
@@ -327,6 +328,45 @@ async function handlePressKey(wv: BrowserWebview, params: Record<string, any>):
|
||||
return { text: `Pressed ${rawKey}` };
|
||||
}
|
||||
|
||||
// Click at a viewport coordinate (percent of the view's width/height) with a
|
||||
// real, trusted CDP mouse event, NO DOM element required. This is what lets the
|
||||
// app agent operate a bare <canvas> game the way a person taps the screen: the
|
||||
// AX-tree click paths (click_index/click_by_name) can't target a canvas because
|
||||
// it exposes no nodes, but a coordinate dispatch lands anywhere. Optional
|
||||
// hold_ms presses and holds (platformers, charge-up mechanics).
|
||||
async function handleClickPoint(wv: BrowserWebview, params: Record<string, any>): Promise<Record<string, any>> {
|
||||
const xPercent = Number(params.xPercent);
|
||||
const yPercent = Number(params.yPercent);
|
||||
if (!Number.isFinite(xPercent) || !Number.isFinite(yPercent)) {
|
||||
return { error: 'xPercent and yPercent are required (0-100, percent of the view).' };
|
||||
}
|
||||
const cx = Math.max(0, Math.min(100, xPercent));
|
||||
const cy = Math.max(0, Math.min(100, yPercent));
|
||||
const button = params.button === 'right' ? 'right' : params.button === 'middle' ? 'middle' : 'left';
|
||||
const holdMs = Math.max(0, Math.min(Number(params.hold_ms) || 0, 5000));
|
||||
// Read the guest's own viewport so coords are correct under zoom/DPR, not the
|
||||
// host element's box. One cheap round-trip; falls back to the element box.
|
||||
let vw = wv.clientWidth, vh = wv.clientHeight;
|
||||
try {
|
||||
const d = await wv.executeJavaScript('({w: window.innerWidth, h: window.innerHeight})');
|
||||
if (d && d.w > 0 && d.h > 0) { vw = d.w; vh = d.h; }
|
||||
} catch { /* use the element box as a fallback */ }
|
||||
const x = (cx / 100) * vw;
|
||||
const y = (cy / 100) * vh;
|
||||
try {
|
||||
await sendCdp(wv, 'Input.dispatchMouseEvent', { type: 'mouseMoved', x, y });
|
||||
await sendCdp(wv, 'Input.dispatchMouseEvent', { type: 'mousePressed', x, y, button, clickCount: 1 });
|
||||
if (holdMs > 0) await new Promise((r) => setTimeout(r, holdMs));
|
||||
await sendCdp(wv, 'Input.dispatchMouseEvent', { type: 'mouseReleased', x, y, button, clickCount: 1 });
|
||||
} catch (err: any) {
|
||||
return { error: `Click point failed: ${err?.message || String(err)}` };
|
||||
}
|
||||
return {
|
||||
text: `Clicked at (${Math.round(x)}, ${Math.round(y)})${holdMs ? ` held ${holdMs}ms` : ''}.`,
|
||||
clickX: cx, clickY: cy, url: wv.getURL(),
|
||||
};
|
||||
}
|
||||
|
||||
// CDP Accessibility.getFullAXTree sees computed roles/names even on hostile sites with unlabeled DOMs.
|
||||
const INTERACTIVE_ROLES = new Set([
|
||||
'button', 'link', 'textbox', 'combobox', 'checkbox', 'menuitem',
|
||||
@@ -941,11 +981,12 @@ const MAX_BATCH_ACTIONS = 5;
|
||||
|
||||
type SubActionType =
|
||||
| 'click_index' | 'press_key' | 'type' | 'wait'
|
||||
| 'scroll' | 'navigate' | 'click' | 'list_interactives';
|
||||
| 'scroll' | 'navigate' | 'click' | 'click_point' | 'list_interactives';
|
||||
|
||||
const BATCH_DISPATCH: Record<SubActionType, (wv: BrowserWebview, p: Record<string, any>) => Promise<Record<string, any>>> = {
|
||||
click_index: handleClickIndex,
|
||||
press_key: handlePressKey,
|
||||
click_point: handleClickPoint,
|
||||
type: handleType,
|
||||
wait: handleWait,
|
||||
scroll: handleScroll,
|
||||
@@ -1298,8 +1339,13 @@ async function handleReplayRoute(wv: BrowserWebview, params: Record<string, any>
|
||||
async function handleEvaluate(wv: BrowserWebview, params: Record<string, any>): Promise<Record<string, any>> {
|
||||
const expression = params.expression as string;
|
||||
if (!expression) return { error: 'expression parameter is required' };
|
||||
// [app-agent] step trace: surface what the app's bridge actually returned.
|
||||
const _bridgeCall = expression.includes('OPENSWARM_APP');
|
||||
try {
|
||||
const result = await wv.executeJavaScript(expression);
|
||||
if (_bridgeCall) {
|
||||
console.log(`[app-agent] BRIDGE eval -> ${typeof result === 'string' ? result : JSON.stringify(result)}`);
|
||||
}
|
||||
const text = typeof result === 'string' ? result : JSON.stringify(result, null, 2);
|
||||
// evaluate is the agent's main read path; sample routes here too (XHRs have fired by now) so the backend can surface the fast network tier once.
|
||||
const routes_available = await countSafeRoutes(wv);
|
||||
@@ -1359,14 +1405,27 @@ async function runBrowserCommand(
|
||||
request_id: string, action: string, browser_id: string, tab_id: string | undefined,
|
||||
params: Record<string, any>,
|
||||
) {
|
||||
// [app-agent] step trace: only for app targets so browser-agent runs stay quiet.
|
||||
const _appTrace = String(browser_id || '').startsWith('app:');
|
||||
if (_appTrace) {
|
||||
console.log(`[app-agent] CMD recv: action=${action} browser_id=${browser_id} tab_id=${tab_id ?? ''} registered=[${registeredKeys().join(', ')}]`);
|
||||
}
|
||||
const wv = await awaitWebview(browser_id, tab_id || undefined);
|
||||
if (!wv) {
|
||||
if (_appTrace) {
|
||||
console.warn(`[app-agent] LOOKUP MISS: no webview for '${browser_id}' (registered keys: [${registeredKeys().join(', ')}]) -> the app card isn't mounted on the active dashboard`);
|
||||
}
|
||||
dashboardWs.send('browser:result', {
|
||||
request_id,
|
||||
error: `Browser card '${browser_id}'${tab_id ? ` tab '${tab_id}'` : ''} not found or not an Electron webview`,
|
||||
});
|
||||
return;
|
||||
}
|
||||
if (_appTrace) {
|
||||
let _url = '';
|
||||
try { _url = wv.getURL(); } catch (_e) {}
|
||||
console.log(`[app-agent] LOOKUP HIT: webview for '${browser_id}' found, url=${_url} loading=${(() => { try { return wv.isLoading(); } catch { return '?'; } })()}`);
|
||||
}
|
||||
|
||||
const detail = params.url || params.selector || params.expression || undefined;
|
||||
setActivity(browser_id, { action: action as BrowserAction, detail });
|
||||
@@ -1427,6 +1486,16 @@ async function runBrowserCommand(
|
||||
});
|
||||
}
|
||||
break;
|
||||
case 'click_point':
|
||||
result = await handleClickPoint(wv, params);
|
||||
if (result.clickX != null && result.clickY != null) {
|
||||
setActivity(browser_id, {
|
||||
action: 'click_point',
|
||||
detail,
|
||||
coords: { xPercent: result.clickX, yPercent: result.clickY },
|
||||
});
|
||||
}
|
||||
break;
|
||||
case 'batch':
|
||||
result = await handleBatch(wv, params);
|
||||
break;
|
||||
|
||||
@@ -50,6 +50,11 @@ export function setActiveTab(browserId: string, tabId: string): void {
|
||||
activeTabMap.set(browserId, tabId);
|
||||
}
|
||||
|
||||
// [app-agent] diagnostic: list currently-registered keys ("browserId:tabId").
|
||||
export function registeredKeys(): string[] {
|
||||
return Array.from(registry.keys());
|
||||
}
|
||||
|
||||
export function getWebview(browserId: string, tabId?: string): BrowserWebview | undefined {
|
||||
const resolvedTabId = tabId || activeTabMap.get(browserId);
|
||||
if (!resolvedTabId) return undefined;
|
||||
|
||||
Reference in New Issue
Block a user