mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-10-01 22:14:51 +02:00
* [haik]: ckpt, added in the test runner skelton i made in another repo -> still gotta make tweaks to fold it onto the current repo * [haik]: restructure backend test layout: move 37 test files from backend/tests/ into backend/tests/test_cases/, add backend/tests/run.sh one-command launcher that auto-provisions both runner and test venvs with stamp-based caching, update runner config.json to point test_paths at test_cases/ and fix venv_python path, revise runner README with run.sh usage and clearer setup instructions, and gitignore the .runner-venv directory * [haik]: refactor: reorganize backend/tests/test_cases from flat structure into domain subdirectories — moved 36 test files into auth/, browser/, labeling/, service/, settings/, and web_search/ for better discoverability and grouping * [haik]: overhaul test picker UI and add post-run rerun loop: replace checkbox glyphs with fzf-style row recolouring (coral=full, lighter coral=partial) and a right-pinned selection dot, add config-driven icon tiers (nerd/emoji/unicode/ascii) with per-glyph graceful degradation, replace the modal -k keyword screen with an inline live-filtering search bar that prunes the tree on every keystroke, add warm Anthropic-dark coral theme, toolbar flag chips replacing the old status line, and a floating help badge overlay; add rerun_prompt.py with inline Textual pill prompt (rerun all/failed/passed/exit) shown after each TTY run, wire it into main.py as a post-run loop; change run_tests to return (exit_code, RunSummary) tracking collected/passed/failed node IDs via new Dashboard methods; add icons field to config.py and config.json (set to nerd), document icon tiers and Nerd Font setup in README, set Hack Nerd Font in .vscode/settings.json, add .runner-venv to linter excludes
53 lines
2.0 KiB
Python
53 lines
2.0 KiB
Python
"""Hot-path waste removals in the browser sub-agent loop.
|
|
|
|
Two per-action costs that were pure waste:
|
|
1. browser_metrics._metrics_dir() ran os.makedirs() on EVERY tool call.
|
|
2. The loop-detection hash serialized a tool's full result (a ~1MB screenshot
|
|
or 15KB read) even for tools that are excluded from loop detection, where
|
|
_detect_loop ignores the hash entirely. These pin both fixes.
|
|
"""
|
|
|
|
import os
|
|
|
|
import backend.apps.agents.browser.browser_metrics as M
|
|
from backend.apps.agents.browser.browser_loop import (
|
|
detect_loop,
|
|
LOOP_DETECTION_EXCLUDED_TOOLS,
|
|
)
|
|
|
|
|
|
def test_metrics_dir_is_cached_makedirs_runs_once(monkeypatch):
|
|
M.P_METRICS_DIR_CACHE = None
|
|
calls = {"n": 0}
|
|
real = os.makedirs
|
|
|
|
def counting(*a, **k):
|
|
calls["n"] += 1
|
|
return real(*a, **k)
|
|
|
|
monkeypatch.setattr(os, "makedirs", counting)
|
|
d1 = M.metrics_dir()
|
|
d2 = M.metrics_dir()
|
|
d3 = M.metrics_dir()
|
|
assert d1 == d2 == d3
|
|
assert calls["n"] == 1, f"makedirs must run once, ran {calls['n']}x"
|
|
|
|
|
|
def test_excluded_tools_never_register_a_loop():
|
|
# The invariant the hash-skip relies on: for every excluded tool, even ten
|
|
# identical calls in a row are NOT a loop, so computing/storing the hash for
|
|
# them was dead work. Setting is_loop=False directly is therefore equivalent.
|
|
for tool in LOOP_DETECTION_EXCLUDED_TOOLS:
|
|
key = (tool, "in", "out")
|
|
assert detect_loop([key] * 10, key) is False, f"{tool} wrongly looped"
|
|
|
|
|
|
def test_non_excluded_tool_still_loops_after_threshold():
|
|
# Guard the other side: the fix must NOT disable loop detection for the tools
|
|
# that need it (clicks/types/etc.).
|
|
key = ("BrowserClick", '{"selector":"#x"}', "clicked")
|
|
# below threshold -> not a loop; at/over threshold within the window -> loop
|
|
assert detect_loop([], key) is False # 1st occurrence: not yet a wall
|
|
assert detect_loop([key], key) is True # 2nd identical (threshold=2): a wall
|
|
assert detect_loop([key] * 5, key) is True
|