diff --git a/e2e/browser-v3/COMPETITIVE_2026-08-08.md b/e2e/browser-v3/COMPETITIVE_2026-08-08.md index 9c147938..61f01c30 100644 --- a/e2e/browser-v3/COMPETITIVE_2026-08-08.md +++ b/e2e/browser-v3/COMPETITIVE_2026-08-08.md @@ -245,3 +245,54 @@ A CMU-adjacent paper is titled *"An Illusion of Progress? Assessing the Current - The anon suite is one I built and then optimised against; the holdout is the guard, and it held. - agent-browser was measured on perception+fill only, not on end-to-end task completion, because it has no agent loop to compare. + +--- + +## 10. BrowserGym / MiniWoB: the first externally-scored number in this whole exercise + +Every other measurement here, ours and theirs, was scored by a harness I wrote. MiniWoB's reward +comes from the task definition, so it cannot be flattered by the person running it. That makes it the +only ground truth in this document. + +Setup that works (the pypi install fails on py3.13 -- greenlet 3.0.3 will not build): + +```bash +uv venv bg-venv --python 3.12 && uv pip install --python bg-venv/bin/python browsergym +git clone --depth 1 https://github.com/Farama-Foundation/miniwob-plusplus +(cd miniwob-plusplus/miniwob/html && python3 -m http.server 8099 &) +MINIWOB_URL=http://localhost:8099/miniwob/ bg-venv/bin/python e2e/browser-v3/miniwob_baseline.py +``` + +341 environments register: **125 MiniWoB**, 215 AssistantBench, 0 WebArena (WebArena needs its own +self-hosted sites). + +### The baseline that reframes everything + +A **~20-line deterministic heuristic with no model at all** -- read the flattened accessibility tree, +match the quoted target in the goal, click or fill it: + +| result | value | +| --- | --- | +| **solved** | **5/15 = 33%** | +| median wall | **1.44s** | +| LLM calls | **0** | +| cost | **$0** | + +Solved: click-test, click-button, click-dialog, focus-text, click-tab (all in 1 step, ~1.2-1.4s). +Failed: click-link, click-checkboxes, enter-text, enter-text-dynamic, focus-text-2, enter-password, +navigate-tree, click-option, simple-algebra, login-user. + +**Why this matters more than any competitor number in this document:** a third of these tasks fall to +twenty lines of code, no model, in 1.4 seconds. Published agent scores on MiniWoB sit in the 70-90% +range. So the honest question for any model-driven browser agent -- ours included -- is not "does it +work" but "what does it buy over the free deterministic baseline, and is that worth 10-40x the +latency and the per-token cost". + +Our own scripted path answers that well (9.8s, 100% where it fires). Our model loop at 37.8s and +browser-use at 35.7s both need to justify ~25x the wall clock of the heuristic. + +### Next step for this harness + +`miniwob_baseline.py` takes task names as argv, so the same 15 (or all 125) can be run against a real +agent policy for a like-for-like number on ground truth someone else owns. That is the missing +measurement in this whole session and it is now one script away. diff --git a/e2e/browser-v3/miniwob_baseline.py b/e2e/browser-v3/miniwob_baseline.py new file mode 100644 index 00000000..633c3b14 --- /dev/null +++ b/e2e/browser-v3/miniwob_baseline.py @@ -0,0 +1,94 @@ +"""MiniWoB via BrowserGym, scored by the benchmark itself rather than by anything I wrote. + +This is the piece every other measurement in this session lacked: ground truth someone else defined. +Our own suites are ones I built and then tuned against; MiniWoB's reward comes from the task, so a +number here cannot be flattered by the harness author. + +The policy under test is deliberately trivial -- a deterministic accessibility-tree heuristic with NO +model: read the flattened axtree, pick the element the goal names, click or fill it. The point is to +establish what the FREE, no-LLM baseline scores, because that is the bar any model-driven agent has +to beat to justify its latency and cost. +""" +import re +import sys +import time + +import gymnasium as gym +import browsergym.miniwob # noqa: F401 registers the envs +from browsergym.utils.obs import flatten_axtree_to_str + +TASKS = sys.argv[1:] or [ + "click-test", "click-button", "click-link", "click-checkboxes", "click-dialog", + "enter-text", "enter-text-dynamic", "focus-text", "focus-text-2", "enter-password", + "click-tab", "navigate-tree", "click-option", "simple-algebra", "login-user", +] + + +def act(ax: str, goal: str): + """One deterministic step from the axtree. No model, no learning, ~20 lines.""" + goal_l = goal.lower() + # A quoted target in the goal names the element outright; prefer an exact label match on it. + want = re.findall(r'"([^"]+)"', goal) + re.findall(r"'([^']+)'", goal) + rows = re.findall(r"\[(\d+)\]\s+(\w+)\s*'([^']*)'", ax) + if not rows: + return None + if want: + for bid, role, name in rows: + if any(w.lower() == name.lower() for w in want): + return f'click("{bid}")' + # Typing goals: fill the first textbox with the quoted payload, then submit if one is offered. + if any(k in goal_l for k in ("enter", "type", "text", "password")): + for bid, role, name in rows: + if role in ("textbox", "searchbox"): + payload = want[0] if want else "hello" + return f'fill("{bid}", "{payload}")' + for bid, role, name in rows: + if role in ("button", "link", "checkbox", "tab", "option"): + return f'click("{bid}")' + return f'click("{rows[0][0]}")' + + +def main() -> None: + wins = 0 + total = 0 + walls = [] + print(f"{'task':22s}{'reward':>8}{'steps':>7}{'wall_s':>9}") + for t in TASKS: + try: + env = gym.make(f"browsergym/miniwob.{t}", headless=True, max_episode_steps=8) + except Exception as e: + print(f"{t:22s}{'ENV ERR':>8} {type(e).__name__}") + continue + t0 = time.time() + reward, steps = 0.0, 0 + try: + obs, _ = env.reset(seed=42) + goal = str(obs.get("goal") or "") + for steps in range(1, 9): + a = act(flatten_axtree_to_str(obs["axtree_object"]), goal) + if not a: + break + obs, r, term, trunc, _ = env.step(a) + reward = max(reward, float(r or 0)) + if term or trunc: + break + except Exception as e: + print(f"{t:22s}{'RUN ERR':>8} {type(e).__name__}: {str(e)[:40]}") + try: + env.close() + except Exception: + pass + continue + w = time.time() - t0 + walls.append(w) + total += 1 + wins += 1 if reward > 0 else 0 + print(f"{t:22s}{reward:>8.2f}{steps:>7d}{w:>9.2f}") + env.close() + import statistics + print(f"\nno-LLM axtree heuristic: {wins}/{total} = {100*wins/total if total else 0:.0f}%" + f" median wall {statistics.median(walls):.2f}s" if walls else "no runs") + + +if __name__ == "__main__": + main()