mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-28 04:24:51 +02:00
Ground truth: one CompWoB page (enter-text-second family) is broken upstream (clears an input before creating it); isolate mode took remaining[:1] every round, so the supervisor retried that one impossible page 201 straight times and 72 healthy tasks starved unseen behind it. browser-use never received them. Supervisor now rotates through remaining tasks and blacklists any that fail setup three times; their fair completion run is live. Score one more for verify-the-verifier: both prior narratives about their CompWoB failures were wrong, and the record now says so. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WsbS5x2rYsMDxP2kW3qqmQ
61 lines
2.8 KiB
Python
61 lines
2.8 KiB
Python
"""Register CompWoB (compositional MiniWoB, Furuta et al.) as BrowserGym tasks.
|
|
|
|
CompWoB exists to test exactly one thing: whether high MiniWoB scores are memorization or ability
|
|
-- specialist systems fell 95%->61% on it. We reuse BrowserGym's own MiniWoB task class untouched,
|
|
pointed at the composed pages, so there is ZERO new scoring code here to be wrong: the reward is
|
|
the same page-owned WOB_REWARD_GLOBAL machinery MiniWoB itself uses, already canary-validated.
|
|
|
|
Task ids: `compwob.<page-name>` -- the runner's existing dotted-id path handles them unchanged.
|
|
Serving: the composed pages live at MINIWOB_URL/../compwob/<name>.html next to the shared assets.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from pathlib import Path
|
|
|
|
from browsergym.core.registration import register_task
|
|
from browsergym.miniwob.base import AbstractMiniwobTask
|
|
|
|
COMPWOB_DIR = Path(os.environ.get(
|
|
"COMPWOB_HTML_DIR", Path.home() / ".cache" / "arena" / "miniwob-legacy" / "html" / "compwob"))
|
|
|
|
|
|
def compwob_page_names() -> list[str]:
|
|
"""Discovered from the served directory, so the registry can never drift from reality."""
|
|
if not COMPWOB_DIR.is_dir():
|
|
return []
|
|
return sorted(p.stem for p in COMPWOB_DIR.glob("*.html"))
|
|
|
|
|
|
ALL_COMPWOB_TASKS: list[type] = []
|
|
|
|
for _name in compwob_page_names():
|
|
# Plain subdomain + a base_url that matches the browser's own URL LITERALLY: validate()
|
|
# string-compares page.url to base_url+subdomain+'.html', so any '../' cleverness terminates
|
|
# every episode with 'invalid url' after its first step (measured). Boring URLs are robust URLs.
|
|
_cls = type(
|
|
f"Compwob_{_name.replace('-', '_').replace('.', '_')}",
|
|
(AbstractMiniwobTask,),
|
|
{"subdomain": _name, "desc": f"CompWoB composed task {_name}"},
|
|
)
|
|
_orig_init = _cls.__init__
|
|
def _init(self, seed, base_url=None, _o=_orig_init, **kw):
|
|
_o(self, seed=seed, base_url=os.environ.get("COMPWOB_URL", "http://localhost:8098/compwob/"), **kw)
|
|
_cls.__init__ = _init
|
|
_orig_setup = _cls.setup
|
|
def _setup(self, page, _o=_orig_setup):
|
|
# genProblem crashes if it runs before the page's own scripts finish attaching (a race the
|
|
# newer chromium loses deterministically). Gate on the page being genuinely ready first.
|
|
try:
|
|
page.wait_for_load_state("networkidle", timeout=15000)
|
|
page.wait_for_function("typeof genProblem === 'function' && !!document.getElementById('wrap')",
|
|
timeout=15000)
|
|
except Exception:
|
|
pass
|
|
return _o(self, page)
|
|
_cls.setup = _setup
|
|
# Stable public id: compwob.<name> (the subdomain's ../ prefix stays an URL detail).
|
|
_cls.get_task_id = classmethod(lambda cls, n=_name: f"compwob.{n}")
|
|
ALL_COMPWOB_TASKS.append(_cls)
|
|
register_task(_cls.get_task_id(), _cls, nondeterministic=_cls.nondeterministic)
|