mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-10-01 22:14:51 +02:00
* [haik]: ckpt, added in the test runner skelton i made in another repo -> still gotta make tweaks to fold it onto the current repo * [haik]: restructure backend test layout: move 37 test files from backend/tests/ into backend/tests/test_cases/, add backend/tests/run.sh one-command launcher that auto-provisions both runner and test venvs with stamp-based caching, update runner config.json to point test_paths at test_cases/ and fix venv_python path, revise runner README with run.sh usage and clearer setup instructions, and gitignore the .runner-venv directory * [haik]: refactor: reorganize backend/tests/test_cases from flat structure into domain subdirectories — moved 36 test files into auth/, browser/, labeling/, service/, settings/, and web_search/ for better discoverability and grouping * [haik]: overhaul test picker UI and add post-run rerun loop: replace checkbox glyphs with fzf-style row recolouring (coral=full, lighter coral=partial) and a right-pinned selection dot, add config-driven icon tiers (nerd/emoji/unicode/ascii) with per-glyph graceful degradation, replace the modal -k keyword screen with an inline live-filtering search bar that prunes the tree on every keystroke, add warm Anthropic-dark coral theme, toolbar flag chips replacing the old status line, and a floating help badge overlay; add rerun_prompt.py with inline Textual pill prompt (rerun all/failed/passed/exit) shown after each TTY run, wire it into main.py as a post-run loop; change run_tests to return (exit_code, RunSummary) tracking collected/passed/failed node IDs via new Dashboard methods; add icons field to config.py and config.json (set to nerd), document icon tiers and Nerd Font setup in README, set Hack Nerd Font in .vscode/settings.json, add .runner-venv to linter excludes
72 lines
2.4 KiB
Python
72 lines
2.4 KiB
Python
"""Ghost-success detector (scripts/analyze-browser-metrics.py): a 'completed'
|
|
task with no verifiable work must be flagged, but honest reads/actions must not.
|
|
This is the guard against features that fail silently without erroring."""
|
|
|
|
import importlib.util
|
|
import os
|
|
|
|
SCRIPT = os.path.join(os.path.dirname(__file__), "..", "..", "scripts", "analyze-browser-metrics.py")
|
|
|
|
|
|
def load():
|
|
spec = importlib.util.spec_from_file_location("abm", SCRIPT)
|
|
m = importlib.util.module_from_spec(spec)
|
|
spec.loader.exec_module(m)
|
|
return m
|
|
|
|
|
|
def ev(tool, ok=True, loop=False, result_len=50):
|
|
return {"tool": tool, "ok": ok, "is_loop": loop, "result_len": result_len}
|
|
|
|
|
|
def test_zero_tools_is_ghost():
|
|
gv = load().ghost_verdict
|
|
is_ghost, reasons = gv({"status": "completed"}, [])
|
|
assert is_ghost and any("ZERO" in r for r in reasons)
|
|
|
|
|
|
def test_read_with_content_is_not_ghost():
|
|
# a pure read task that actually returned data is legitimate work
|
|
gv = load().ghost_verdict
|
|
is_ghost, _ = gv({"status": "completed"}, [ev("BrowserGetText", result_len=120)])
|
|
assert not is_ghost
|
|
|
|
|
|
def test_empty_reads_only_is_ghost():
|
|
gv = load().ghost_verdict
|
|
is_ghost, reasons = gv({"status": "completed"}, [
|
|
ev("BrowserGetText", result_len=0), ev("BrowserScreenshot", result_len=0)])
|
|
assert is_ghost
|
|
|
|
|
|
def test_all_productive_actions_errored_is_ghost():
|
|
gv = load().ghost_verdict
|
|
is_ghost, _ = gv({"status": "completed"}, [
|
|
ev("BrowserClickIndex", ok=False), ev("BrowserClickIndex", ok=False)])
|
|
assert is_ghost
|
|
|
|
|
|
def test_honest_action_is_not_ghost():
|
|
gv = load().ghost_verdict
|
|
is_ghost, _ = gv({"status": "completed"}, [ev("BrowserNavigate"), ev("BrowserClickIndex")])
|
|
assert not is_ghost
|
|
|
|
|
|
def test_loop_during_completed_is_ghost():
|
|
gv = load().ghost_verdict
|
|
is_ghost, reasons = gv({"status": "completed"}, [ev("BrowserNavigate"), ev("BrowserClickIndex", loop=True)])
|
|
assert is_ghost and any("loop" in r.lower() for r in reasons)
|
|
|
|
|
|
def test_errored_task_is_never_ghost():
|
|
# an honest failure (status=error) is not a ghost; ghosts are fake successes
|
|
gv = load().ghost_verdict
|
|
assert not gv({"status": "error"}, [])[0]
|
|
|
|
|
|
def test_tier_mapping_present():
|
|
# the analyzer's _PRODUCTIVE set must include the real mutation tools
|
|
m = load()
|
|
for t in ("BrowserClick", "BrowserClickIndex", "BrowserType", "BrowserNavigate", "BrowserReplayRoute"):
|
|
assert t in m._PRODUCTIVE
|