diff --git a/e2e/browser-v3/arena/llm_policy.py b/e2e/browser-v3/arena/llm_policy.py index ff83b13e..80db3f03 100644 --- a/e2e/browser-v3/arena/llm_policy.py +++ b/e2e/browser-v3/arena/llm_policy.py @@ -40,6 +40,9 @@ clear(index) | hover(index) | focus(index) | scroll(dx, dy) | drag_and_drop(from For targets with no element of their own (a spot on a canvas, a slider position), use coordinates: mouse_click(x, y) | mouse_dblclick(x, y) | mouse_drag_and_drop(from_x, from_y, to_x, to_y) For a combobox/listbox row showing options="...", pick with select_option(index, "exact option"). +A row marked `off` is below the fold -- acting on it scrolls it into view automatically, so use it +directly rather than scrolling blindly. `#k/n` is the row's position inside its repeated group, so +"the 2nd result/post/story" means the row marked #2/n. Prefer filling the box the goal names, then submitting. Do not repeat an action that already worked. If an action did not change the page, try a DIFFERENT action, never the same one again.""" @@ -198,7 +201,8 @@ class OpenSwarmLlmPolicy(LlmPolicy): def view(self, obs: dict[str, Any], goal: str) -> tuple[str, int]: raw_items: list[RankItem] = perception.interactives( - obs, include_clickable=self.clickable, attr_hints=self.hints) + obs, include_clickable=self.clickable, attr_hints=self.hints, + include_offscreen=self.offscreen) shown, truncated = rank_and_cap(raw_items, goal=goal) new = {it.bid for it in shown} - self.prev_bids if self.prev_bids else set() self.prev_bids = {it.bid for it in shown} @@ -344,6 +348,9 @@ class OpenSwarmLlmPolicy(LlmPolicy): row_values: dict[int, str] = field(default_factory=dict) + # v19: include below-the-fold rows (the action layer scrolls them into view). + offscreen: bool = False + # v18: ensemble dispatcher -- ONE agent, ONE episode, but the rung set is chosen per task from # page/goal FEATURES at first sight (never task names). The systematic cross-version wins were # mode-shaped: forms want strict per-field verification, canvases want eyes every turn, consoles @@ -697,12 +704,13 @@ def build(name: str, model: str = "", endpoint: str = "", **_: Any) -> Any: scripted_drag=True, auto_complete=True, som=False, native_pickers=True, verify_terminal=True, post_mouse_vision=True, multi_cap=6, fill_verify=True, **v17) - if name == "osw-llm-v18": # v17 + feature-dispatched episode modes (form/geometry/console/game) + if name in ("osw-llm-v18", "osw-llm-v19"): # v18 + (v19) off-screen rows and group ordinals v18 = dict(v7, system=OSW_SYSTEM_V8 + OSW_SYSTEM_V9_WIDGETS + OSW_SYSTEM_V16, max_tokens=500) return OpenSwarmLlmPolicy(name=name, multi=True, vision="progressive", fastpath=True, scripted_drag=True, auto_complete=True, som=False, native_pickers=True, verify_terminal=True, post_mouse_vision=True, - multi_cap=6, fill_verify=True, dispatch=True, **v18) + multi_cap=6, fill_verify=True, dispatch=True, + offscreen=(name == "osw-llm-v19"), **v18) if name == "osw-llm-v16": # v15 + verify-terminal + look-act-look + rapid-fire cap v16 = dict(v7, system=OSW_SYSTEM_V8 + OSW_SYSTEM_V9_WIDGETS + OSW_SYSTEM_V16, max_tokens=500) return OpenSwarmLlmPolicy(name=name, multi=True, vision="progressive", fastpath=True, diff --git a/e2e/browser-v3/arena/perception.py b/e2e/browser-v3/arena/perception.py index 74cb7827..cc2f57dc 100644 --- a/e2e/browser-v3/arena/perception.py +++ b/e2e/browser-v3/arena/perception.py @@ -72,7 +72,8 @@ def build_context(nodes: list[dict[str, Any]], by_id: dict[str, dict[str, Any]], def interactives(obs: dict[str, Any], include_hidden: bool = False, - include_clickable: bool = False, attr_hints: bool = False) -> list[RankItem]: + include_clickable: bool = False, attr_hints: bool = False, + include_offscreen: bool = False) -> list[RankItem]: """Every actionable node in document order, before any ranking or capping is applied. include_clickable is the technique ingested from browser-use: elements the page wires for @@ -98,8 +99,13 @@ def interactives(obs: dict[str, Any], include_hidden: bool = False, and bool((extra.get(str(bid)) or {}).get("clickable"))) if not (is_role or is_clickable): continue - if not include_hidden and not visible(str(bid), extra): - continue + props = extra.get(str(bid)) or {} + onscreen = visible(str(bid), extra) + # A row with a bbox is RENDERED; visibility is unactionable. options: list[str] | None = None + # Rendered but outside the viewport. The action layer scrolls into view automatically, so these + # are perfectly actionable -- hiding them made every long feed unsolvable (the @ashlea row the + # goal named was simply absent from the menu). + offscreen: bool = False def role_priority(role: str) -> int: @@ -109,6 +113,29 @@ def render(items: list[RankItem], truncated: int, new_bids: set[str] | None = No for el in items: k = f"{el.role}|{el.name}" counts[k] = counts.get(k, 0) + 1 + # Ordinals inside repeated structures: search results, feed posts, product cards. Counting is + # what goals mean by "the 2nd result"; without it the model guesses among interleaved links. + ordinals: dict[int, str] = {} + by_name: dict[str, list[int]] = {} + for i, el in enumerate(items, 1): + by_name.setdefault(f"{el.role}|{el.name}", []).append(i) + for key, idxs in by_name.items(): + if len(idxs) > 1: + for k, i in enumerate(idxs, 1): + ordinals[i] = f" #{k}/{len(idxs)}" + run_start, run_role = 0, None + runs: list[tuple[int, int, str]] = [] + for i, el in enumerate(items, 1): + if el.role != run_role: + if run_role is not None and i - run_start >= 3: + runs.append((run_start, i - 1, run_role)) + run_start, run_role = i, el.role + if run_role is not None and len(items) + 1 - run_start >= 3: + runs.append((run_start, len(items), run_role)) + for a, b, _role in runs: + for k, i in enumerate(range(a, b + 1), 1): + ordinals.setdefault(i, f" #{k}/{b - a + 1}") + lines = [] for i, el in enumerate(items, 1): dup = counts.get(f"{el.role}|{el.name}", 0) > 1 @@ -117,6 +144,7 @@ def render(items: list[RankItem], truncated: int, new_bids: set[str] | None = No ctx = f' ctx="{el.context}"' if (dup or weak_name) and el.context else "" val = f' value="{el.value}"' if el.value else "" star = "*" if new_bids and el.bid in new_bids else "" + off = " off" if el.offscreen else "" # Coordinates whenever the label alone cannot pin the row; a named button needs no geometry. pos = f" center=({el.center[0]:.0f},{el.center[1]:.0f})" if el.center and weak_name else "" opts = "" @@ -124,7 +152,8 @@ def render(items: list[RankItem], truncated: int, new_bids: set[str] | None = No shown_opts = el.options[:12] more = f" +{len(el.options) - 12}" if len(el.options) > 12 else "" opts = f' options="{"|".join(shown_opts)}{more}"' - lines.append(f'[{i}]{star}<{el.role} "{el.name}"{ctx}{val}{pos}{opts}>') + ordinal = ordinals.get(i, "") + lines.append(f'[{i}]{star}<{el.role} "{el.name}"{ctx}{val}{pos}{opts}{ordinal}{off}>') head = f"{len(lines)} interactive elements (* = new since your last look):" text = head + "\n" + "\n".join(lines) if truncated > 0: