Files
openswarm/backend/tests/test_cases/browser/test_browser_hotpath_waste.py
T
haikdcandGitHub 9c95342704 Haik/feat/test runner (#86)
* [haik]: ckpt, added in the test runner skelton i made in another repo -> still gotta make tweaks to fold it onto the current repo

* [haik]: restructure backend test layout: move 37 test files from backend/tests/ into backend/tests/test_cases/, add backend/tests/run.sh one-command launcher that auto-provisions both runner and test venvs with stamp-based caching, update runner config.json to point test_paths at test_cases/ and fix venv_python path, revise runner README with run.sh usage and clearer setup instructions, and gitignore the .runner-venv directory

* [haik]: refactor: reorganize backend/tests/test_cases from flat structure into domain subdirectories — moved 36 test files into auth/, browser/, labeling/, service/, settings/, and web_search/ for better discoverability and grouping

* [haik]: overhaul test picker UI and add post-run rerun loop: replace checkbox glyphs with fzf-style row recolouring (coral=full, lighter coral=partial) and a right-pinned selection dot, add config-driven icon tiers (nerd/emoji/unicode/ascii) with per-glyph graceful degradation, replace the modal -k keyword screen with an inline live-filtering search bar that prunes the tree on every keystroke, add warm Anthropic-dark coral theme, toolbar flag chips replacing the old status line, and a floating help badge overlay; add rerun_prompt.py with inline Textual pill prompt (rerun all/failed/passed/exit) shown after each TTY run, wire it into main.py as a post-run loop; change run_tests to return (exit_code, RunSummary) tracking collected/passed/failed node IDs via new Dashboard methods; add icons field to config.py and config.json (set to nerd), document icon tiers and Nerd Font setup in README, set Hack Nerd Font in .vscode/settings.json, add .runner-venv to linter excludes
2026-06-14 07:02:30 -07:00

53 lines
2.0 KiB
Python

"""Hot-path waste removals in the browser sub-agent loop.
Two per-action costs that were pure waste:
1. browser_metrics._metrics_dir() ran os.makedirs() on EVERY tool call.
2. The loop-detection hash serialized a tool's full result (a ~1MB screenshot
or 15KB read) even for tools that are excluded from loop detection, where
_detect_loop ignores the hash entirely. These pin both fixes.
"""
import os
import backend.apps.agents.browser.browser_metrics as M
from backend.apps.agents.browser.browser_loop import (
detect_loop,
LOOP_DETECTION_EXCLUDED_TOOLS,
)
def test_metrics_dir_is_cached_makedirs_runs_once(monkeypatch):
M.P_METRICS_DIR_CACHE = None
calls = {"n": 0}
real = os.makedirs
def counting(*a, **k):
calls["n"] += 1
return real(*a, **k)
monkeypatch.setattr(os, "makedirs", counting)
d1 = M.metrics_dir()
d2 = M.metrics_dir()
d3 = M.metrics_dir()
assert d1 == d2 == d3
assert calls["n"] == 1, f"makedirs must run once, ran {calls['n']}x"
def test_excluded_tools_never_register_a_loop():
# The invariant the hash-skip relies on: for every excluded tool, even ten
# identical calls in a row are NOT a loop, so computing/storing the hash for
# them was dead work. Setting is_loop=False directly is therefore equivalent.
for tool in LOOP_DETECTION_EXCLUDED_TOOLS:
key = (tool, "in", "out")
assert detect_loop([key] * 10, key) is False, f"{tool} wrongly looped"
def test_non_excluded_tool_still_loops_after_threshold():
# Guard the other side: the fix must NOT disable loop detection for the tools
# that need it (clicks/types/etc.).
key = ("BrowserClick", '{"selector":"#x"}', "clicked")
# below threshold -> not a loop; at/over threshold within the window -> loop
assert detect_loop([], key) is False # 1st occurrence: not yet a wall
assert detect_loop([key], key) is True # 2nd identical (threshold=2): a wall
assert detect_loop([key] * 5, key) is True