From 118fde5623062fd44cb16c02b1fc135ef69fab25 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Sun, 16 Aug 2026 22:26:36 -0700 Subject: [PATCH] arena: atomic-write measurement -- my prediction was WRONG (7%, not high); verified-writes ~7-8% overall is a real capability wall, clause 5 furthest-open; verifier bugs fixed before booking Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01WsbS5x2rYsMDxP2kW3qqmQ --- e2e/browser-v3/arena/ARENA.md | 12 ++++++++++++ e2e/browser-v3/arena/vw_verify.py | 11 ++++++----- 2 files changed, 18 insertions(+), 5 deletions(-) diff --git a/e2e/browser-v3/arena/ARENA.md b/e2e/browser-v3/arena/ARENA.md index 726bff93..6f7435dc 100644 --- a/e2e/browser-v3/arena/ARENA.md +++ b/e2e/browser-v3/arena/ARENA.md @@ -1062,6 +1062,18 @@ multi-step create/edit writes. Scorer caveats: a few url=last tasks need the epi tighten with a cleaned write-only partition, but the signal (far below 95) is unambiguous. Honest verdict: verified-writes is capability-gated for complex writes, not a quick clause to close. +ATOMIC-WRITE MEASUREMENT (2026-08-16, precise): I predicted atomic writes (post-comment / change- +bio / upvote) would score MUCH higher than the 8% complex-write number. **That prediction was +WRONG: 1/15 = 7% (Wilson95 lo 1%).** Upvote tasks fail the vote-class check outright; only a +single MR-comment write verifiably persisted. (Verifier bugs found+fixed first -- func:-prefixed +and url=last evals need episode context and are marked not-independently-checkable rather than +scored 0; a broken instrument must never book a 0, per the twice-learned rule.) CLAUSE 5 HONEST +STATE: verified writes ~7-8% overall, atomic and complex alike -- a REAL capability wall, ~88 +points from the >=95 target, the furthest-open clause we have. The agent navigates to the right +place and often composes the right content, but the write does not persist. Not closable by +scaffolding; needs write-completion capability work (submit-confirm loops, post-write read-back +retries). Booked as OPEN-hard. + ## Online-Mind2Web head-to-head scoping (2026-08-16) — the honest bu-max comparison Runnable in principle: 300 tasks / 136 live sites, CC-BY-4.0, HuggingFace osunlp/Online-Mind2Web. This is where browser-use Cloud claims 97 and open-source SOTA (Avenir-Web) is 53.7 — the only diff --git a/e2e/browser-v3/arena/vw_verify.py b/e2e/browser-v3/arena/vw_verify.py index 6a726d43..ad004d79 100644 --- a/e2e/browser-v3/arena/vw_verify.py +++ b/e2e/browser-v3/arena/vw_verify.py @@ -15,21 +15,22 @@ def check(pg, task) -> tuple[bool, str]: url = c.get("url", "") for k, v in SUB.items(): url = url.replace(k, v) - if url in ("last", ""): - return None, "url=last (needs episode final page, skipped)" + if url in ("last", "") or url.startswith("func:"): + return None, "url needs episode context (last/func:) — not independently checkable" try: pg.goto(url, timeout=15000, wait_until="domcontentloaded"); pg.wait_for_timeout(1200) except Exception as e: return False, f"fetch-fail {str(e)[:40]}" - loc = c.get("locator", "") + loc = c.get("locator") or "" if loc and loc.startswith("document."): try: - text = pg.evaluate(f"() => {{ const e = {loc}; return e ? (e.textContent||e.outerHTML) : ''; }}") + text = pg.evaluate(f"() => {{ try {{ const e = {loc}; return e ? (e.textContent||e.outerHTML||'') : ''; }} catch(x) {{ return ''; }} }}") or "" except Exception: text = pg.content() else: text = pg.content() - req = c.get("required_contents", {}) + req = c.get("required_contents") or {} + text = text or "" if "must_include" in req: if not all(str(x) in text for x in req["must_include"]): return False, "must_include absent"