arena: the "forms" deficit was never forms -- it was viewport-blind perception

Loss traces killed the label: social-media-some fails because the @ashlea row the goal
names sits below the fold and our menu DROPPED every below-fold element, so the agent
scrolled blind; search-engine fails because "the 2nd result" is uncountable when result
links interleave with pagination links. Both are perception bugs with generic fixes:
rendered-but-off-screen rows now appear marked `off` (the action layer scrolls them into
view, so hiding them only blinded us), and rows inside repeated structures carry `#k/n`
group ordinals so "the 2nd result/post/story" is directly addressable. Verified: the
@ashlea retweet button is now row 23 (#6/11 off) where before it was absent entirely.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WsbS5x2rYsMDxP2kW3qqmQ
This commit is contained in:
ciregenz
2026-08-12 10:59:41 -07:00
co-authored by Claude Fable 5
parent 7cc3105a6a
commit 3a34795ba8
3 changed files with 51 additions and 7 deletions
+11 -3
View File
@@ -40,6 +40,9 @@ clear(index) | hover(index) | focus(index) | scroll(dx, dy) | drag_and_drop(from
For targets with no element of their own (a spot on a canvas, a slider position), use coordinates:
mouse_click(x, y) | mouse_dblclick(x, y) | mouse_drag_and_drop(from_x, from_y, to_x, to_y)
For a combobox/listbox row showing options="...", pick with select_option(index, "exact option").
A row marked `off` is below the fold -- acting on it scrolls it into view automatically, so use it
directly rather than scrolling blindly. `#k/n` is the row's position inside its repeated group, so
"the 2nd result/post/story" means the row marked #2/n.
Prefer filling the box the goal names, then submitting. Do not repeat an action that already worked.
If an action did not change the page, try a DIFFERENT action, never the same one again."""
@@ -198,7 +201,8 @@ class OpenSwarmLlmPolicy(LlmPolicy):
def view(self, obs: dict[str, Any], goal: str) -> tuple[str, int]:
raw_items: list[RankItem] = perception.interactives(
obs, include_clickable=self.clickable, attr_hints=self.hints)
obs, include_clickable=self.clickable, attr_hints=self.hints,
include_offscreen=self.offscreen)
shown, truncated = rank_and_cap(raw_items, goal=goal)
new = {it.bid for it in shown} - self.prev_bids if self.prev_bids else set()
self.prev_bids = {it.bid for it in shown}
@@ -344,6 +348,9 @@ class OpenSwarmLlmPolicy(LlmPolicy):
row_values: dict[int, str] = field(default_factory=dict)
# v19: include below-the-fold rows (the action layer scrolls them into view).
offscreen: bool = False
# v18: ensemble dispatcher -- ONE agent, ONE episode, but the rung set is chosen per task from
# page/goal FEATURES at first sight (never task names). The systematic cross-version wins were
# mode-shaped: forms want strict per-field verification, canvases want eyes every turn, consoles
@@ -697,12 +704,13 @@ def build(name: str, model: str = "", endpoint: str = "", **_: Any) -> Any:
scripted_drag=True, auto_complete=True, som=False,
native_pickers=True, verify_terminal=True, post_mouse_vision=True,
multi_cap=6, fill_verify=True, **v17)
if name == "osw-llm-v18": # v17 + feature-dispatched episode modes (form/geometry/console/game)
if name in ("osw-llm-v18", "osw-llm-v19"): # v18 + (v19) off-screen rows and group ordinals
v18 = dict(v7, system=OSW_SYSTEM_V8 + OSW_SYSTEM_V9_WIDGETS + OSW_SYSTEM_V16, max_tokens=500)
return OpenSwarmLlmPolicy(name=name, multi=True, vision="progressive", fastpath=True,
scripted_drag=True, auto_complete=True, som=False,
native_pickers=True, verify_terminal=True, post_mouse_vision=True,
multi_cap=6, fill_verify=True, dispatch=True, **v18)
multi_cap=6, fill_verify=True, dispatch=True,
offscreen=(name == "osw-llm-v19"), **v18)
if name == "osw-llm-v16": # v15 + verify-terminal + look-act-look + rapid-fire cap
v16 = dict(v7, system=OSW_SYSTEM_V8 + OSW_SYSTEM_V9_WIDGETS + OSW_SYSTEM_V16, max_tokens=500)
return OpenSwarmLlmPolicy(name=name, multi=True, vision="progressive", fastpath=True,
+10 -3
View File
@@ -72,7 +72,8 @@ def build_context(nodes: list[dict[str, Any]], by_id: dict[str, dict[str, Any]],
def interactives(obs: dict[str, Any], include_hidden: bool = False,
include_clickable: bool = False, attr_hints: bool = False) -> list[RankItem]:
include_clickable: bool = False, attr_hints: bool = False,
include_offscreen: bool = False) -> list[RankItem]:
"""Every actionable node in document order, before any ranking or capping is applied.
include_clickable is the technique ingested from browser-use: elements the page wires for
@@ -98,8 +99,13 @@ def interactives(obs: dict[str, Any], include_hidden: bool = False,
and bool((extra.get(str(bid)) or {}).get("clickable")))
if not (is_role or is_clickable):
continue
if not include_hidden and not visible(str(bid), extra):
continue
props = extra.get(str(bid)) or {}
onscreen = visible(str(bid), extra)
# A row with a bbox is RENDERED; visibility<threshold just means below the fold. display:none
# yields no bbox and stays excluded.
if not onscreen and not include_hidden:
if not (include_offscreen and props.get("bbox")):
continue
name = node_name(n).strip()
# Nameless rows resolve by their OWN text first ('dignissim' from the child text node), and
# only then by DOM identity ('(trash)'). The class hint alone made every link in a tab panel
@@ -118,6 +124,7 @@ def interactives(obs: dict[str, Any], include_hidden: bool = False,
context=build_context(nodes, by_id, n),
center=center,
options=child_options(by_id, n) if role in ("combobox", "listbox", "menu") else None,
offscreen=not onscreen,
))
return out
+30 -1
View File
@@ -47,6 +47,10 @@ class RankItem:
center: tuple[float, float] | None = None
# Child option labels for combobox/listbox rows; without them a closed <select> is unactionable.
options: list[str] | None = None
# Rendered but outside the viewport. The action layer scrolls into view automatically, so these
# are perfectly actionable -- hiding them made every long feed unsolvable (the @ashlea row the
# goal named was simply absent from the menu).
offscreen: bool = False
def role_priority(role: str) -> int:
@@ -109,6 +113,29 @@ def render(items: list[RankItem], truncated: int, new_bids: set[str] | None = No
for el in items:
k = f"{el.role}|{el.name}"
counts[k] = counts.get(k, 0) + 1
# Ordinals inside repeated structures: search results, feed posts, product cards. Counting is
# what goals mean by "the 2nd result"; without it the model guesses among interleaved links.
ordinals: dict[int, str] = {}
by_name: dict[str, list[int]] = {}
for i, el in enumerate(items, 1):
by_name.setdefault(f"{el.role}|{el.name}", []).append(i)
for key, idxs in by_name.items():
if len(idxs) > 1:
for k, i in enumerate(idxs, 1):
ordinals[i] = f" #{k}/{len(idxs)}"
run_start, run_role = 0, None
runs: list[tuple[int, int, str]] = []
for i, el in enumerate(items, 1):
if el.role != run_role:
if run_role is not None and i - run_start >= 3:
runs.append((run_start, i - 1, run_role))
run_start, run_role = i, el.role
if run_role is not None and len(items) + 1 - run_start >= 3:
runs.append((run_start, len(items), run_role))
for a, b, _role in runs:
for k, i in enumerate(range(a, b + 1), 1):
ordinals.setdefault(i, f" #{k}/{b - a + 1}")
lines = []
for i, el in enumerate(items, 1):
dup = counts.get(f"{el.role}|{el.name}", 0) > 1
@@ -117,6 +144,7 @@ def render(items: list[RankItem], truncated: int, new_bids: set[str] | None = No
ctx = f' ctx="{el.context}"' if (dup or weak_name) and el.context else ""
val = f' value="{el.value}"' if el.value else ""
star = "*" if new_bids and el.bid in new_bids else ""
off = " off" if el.offscreen else ""
# Coordinates whenever the label alone cannot pin the row; a named button needs no geometry.
pos = f" center=({el.center[0]:.0f},{el.center[1]:.0f})" if el.center and weak_name else ""
opts = ""
@@ -124,7 +152,8 @@ def render(items: list[RankItem], truncated: int, new_bids: set[str] | None = No
shown_opts = el.options[:12]
more = f" +{len(el.options) - 12}" if len(el.options) > 12 else ""
opts = f' options="{"|".join(shown_opts)}{more}"'
lines.append(f'[{i}]{star}<{el.role} "{el.name}"{ctx}{val}{pos}{opts}>')
ordinal = ordinals.get(i, "")
lines.append(f'[{i}]{star}<{el.role} "{el.name}"{ctx}{val}{pos}{opts}{ordinal}{off}>')
head = f"{len(lines)} interactive elements (* = new since your last look):"
text = head + "\n" + "\n".join(lines)
if truncated > 0: