Files
openswarm/backend/tests/test_cases/service/test_ghost_detector.py
T
haikdcandGitHub 9c95342704 Haik/feat/test runner (#86)
* [haik]: ckpt, added in the test runner skelton i made in another repo -> still gotta make tweaks to fold it onto the current repo

* [haik]: restructure backend test layout: move 37 test files from backend/tests/ into backend/tests/test_cases/, add backend/tests/run.sh one-command launcher that auto-provisions both runner and test venvs with stamp-based caching, update runner config.json to point test_paths at test_cases/ and fix venv_python path, revise runner README with run.sh usage and clearer setup instructions, and gitignore the .runner-venv directory

* [haik]: refactor: reorganize backend/tests/test_cases from flat structure into domain subdirectories — moved 36 test files into auth/, browser/, labeling/, service/, settings/, and web_search/ for better discoverability and grouping

* [haik]: overhaul test picker UI and add post-run rerun loop: replace checkbox glyphs with fzf-style row recolouring (coral=full, lighter coral=partial) and a right-pinned selection dot, add config-driven icon tiers (nerd/emoji/unicode/ascii) with per-glyph graceful degradation, replace the modal -k keyword screen with an inline live-filtering search bar that prunes the tree on every keystroke, add warm Anthropic-dark coral theme, toolbar flag chips replacing the old status line, and a floating help badge overlay; add rerun_prompt.py with inline Textual pill prompt (rerun all/failed/passed/exit) shown after each TTY run, wire it into main.py as a post-run loop; change run_tests to return (exit_code, RunSummary) tracking collected/passed/failed node IDs via new Dashboard methods; add icons field to config.py and config.json (set to nerd), document icon tiers and Nerd Font setup in README, set Hack Nerd Font in .vscode/settings.json, add .runner-venv to linter excludes
2026-06-14 07:02:30 -07:00

72 lines
2.4 KiB
Python

"""Ghost-success detector (scripts/analyze-browser-metrics.py): a 'completed'
task with no verifiable work must be flagged, but honest reads/actions must not.
This is the guard against features that fail silently without erroring."""
import importlib.util
import os
SCRIPT = os.path.join(os.path.dirname(__file__), "..", "..", "scripts", "analyze-browser-metrics.py")
def load():
spec = importlib.util.spec_from_file_location("abm", SCRIPT)
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)
return m
def ev(tool, ok=True, loop=False, result_len=50):
return {"tool": tool, "ok": ok, "is_loop": loop, "result_len": result_len}
def test_zero_tools_is_ghost():
gv = load().ghost_verdict
is_ghost, reasons = gv({"status": "completed"}, [])
assert is_ghost and any("ZERO" in r for r in reasons)
def test_read_with_content_is_not_ghost():
# a pure read task that actually returned data is legitimate work
gv = load().ghost_verdict
is_ghost, _ = gv({"status": "completed"}, [ev("BrowserGetText", result_len=120)])
assert not is_ghost
def test_empty_reads_only_is_ghost():
gv = load().ghost_verdict
is_ghost, reasons = gv({"status": "completed"}, [
ev("BrowserGetText", result_len=0), ev("BrowserScreenshot", result_len=0)])
assert is_ghost
def test_all_productive_actions_errored_is_ghost():
gv = load().ghost_verdict
is_ghost, _ = gv({"status": "completed"}, [
ev("BrowserClickIndex", ok=False), ev("BrowserClickIndex", ok=False)])
assert is_ghost
def test_honest_action_is_not_ghost():
gv = load().ghost_verdict
is_ghost, _ = gv({"status": "completed"}, [ev("BrowserNavigate"), ev("BrowserClickIndex")])
assert not is_ghost
def test_loop_during_completed_is_ghost():
gv = load().ghost_verdict
is_ghost, reasons = gv({"status": "completed"}, [ev("BrowserNavigate"), ev("BrowserClickIndex", loop=True)])
assert is_ghost and any("loop" in r.lower() for r in reasons)
def test_errored_task_is_never_ghost():
# an honest failure (status=error) is not a ghost; ghosts are fake successes
gv = load().ghost_verdict
assert not gv({"status": "error"}, [])[0]
def test_tier_mapping_present():
# the analyzer's _PRODUCTIVE set must include the real mutation tools
m = load()
for t in ("BrowserClick", "BrowserClickIndex", "BrowserType", "BrowserNavigate", "BrowserReplayRoute"):
assert t in m._PRODUCTIVE