Files
openswarm/backend/tests/test_cases/browser/test_browser_self_audit.py
T
haikdcandGitHub 9c95342704 Haik/feat/test runner (#86)
* [haik]: ckpt, added in the test runner skelton i made in another repo -> still gotta make tweaks to fold it onto the current repo

* [haik]: restructure backend test layout: move 37 test files from backend/tests/ into backend/tests/test_cases/, add backend/tests/run.sh one-command launcher that auto-provisions both runner and test venvs with stamp-based caching, update runner config.json to point test_paths at test_cases/ and fix venv_python path, revise runner README with run.sh usage and clearer setup instructions, and gitignore the .runner-venv directory

* [haik]: refactor: reorganize backend/tests/test_cases from flat structure into domain subdirectories — moved 36 test files into auth/, browser/, labeling/, service/, settings/, and web_search/ for better discoverability and grouping

* [haik]: overhaul test picker UI and add post-run rerun loop: replace checkbox glyphs with fzf-style row recolouring (coral=full, lighter coral=partial) and a right-pinned selection dot, add config-driven icon tiers (nerd/emoji/unicode/ascii) with per-glyph graceful degradation, replace the modal -k keyword screen with an inline live-filtering search bar that prunes the tree on every keystroke, add warm Anthropic-dark coral theme, toolbar flag chips replacing the old status line, and a floating help badge overlay; add rerun_prompt.py with inline Textual pill prompt (rerun all/failed/passed/exit) shown after each TTY run, wire it into main.py as a post-run loop; change run_tests to return (exit_code, RunSummary) tracking collected/passed/failed node IDs via new Dashboard methods; add icons field to config.py and config.json (set to nerd), document icon tiers and Nerd Font setup in README, set Hack Nerd Font in .vscode/settings.json, add .runner-venv to linter excludes
2026-06-14 07:02:30 -07:00

96 lines
4.1 KiB
Python

"""Self-audit (tier 4): reads its own run metrics + skill lifecycle and PROPOSES
fixes for a human, never changes anything. Proves the detectors fire on the real
failure shapes (thrash, stall, error-heavy) and stay quiet on a clean history."""
import json
import os
import tempfile
from backend.apps.agents.browser import browser_self_audit as audit
def write_rows(d, name, rows):
with open(os.path.join(d, name), "w", encoding="utf-8") as f:
for r in rows:
f.write(json.dumps(r) + "\n")
def test_thrash_is_flagged_only_without_a_promote():
d = tempfile.mkdtemp()
# a skill edited/quarantined 4x and NEVER promoted = thrash
ev = [{"kind": "edit", "host": "x.com", "task_sig": "s1"} for _ in range(3)]
ev += [{"kind": "quarantine", "host": "x.com", "task_sig": "s1"}]
# a healthy skill: edited a couple times THEN promoted = not thrash
ev += [{"kind": "edit", "host": "y.com", "task_sig": "s2"},
{"kind": "promote", "host": "y.com", "task_sig": "s2"}]
write_rows(d, "skill_events.jsonl", ev)
write_rows(d, "tasks.jsonl", [])
r = audit.audit(d)
thrash = [f for f in r["findings"] if f["kind"] == "thrash"]
assert len(thrash) == 1 and thrash[0]["host"] == "x.com"
def test_stall_flags_runs_far_above_the_host_norm():
d = tempfile.mkdtemp()
# six fast runs (norm ~4) and two big spikes on the same card
tasks = [{"browser_id": "b1", "turns": n} for n in (4, 4, 5, 3, 4, 4)]
tasks += [{"browser_id": "b1", "turns": 18}, {"browser_id": "b1", "turns": 20}]
write_rows(d, "tasks.jsonl", tasks)
write_rows(d, "skill_events.jsonl", [])
r = audit.audit(d)
assert any(f["kind"] == "stall" for f in r["findings"])
def test_error_rate_flags_a_systemically_failing_host():
d = tempfile.mkdtemp()
tasks = [{"browser_id": "b9", "tool_calls": 30,
"recurring_errors": {"index no longer valid": 12}}]
write_rows(d, "tasks.jsonl", tasks)
write_rows(d, "skill_events.jsonl", [])
r = audit.audit(d)
assert any(f["kind"] == "error_rate" for f in r["findings"])
def test_clean_history_proposes_nothing():
d = tempfile.mkdtemp()
write_rows(d, "tasks.jsonl", [{"browser_id": "b1", "turns": 4, "tool_calls": 5} for _ in range(6)])
write_rows(d, "skill_events.jsonl", [{"kind": "learn", "host": "x.com", "task_sig": "s"},
{"kind": "promote", "host": "x.com", "task_sig": "s"}])
r = audit.audit(d)
assert r["findings"] == []
assert "learning cleanly" in audit.render_report(r)
def test_audit_fires_every_n_finished_tasks(monkeypatch, tmp_path, mocker):
# the trigger refreshes the report once every N tasks, off the hot path. Make
# threads synchronous so the test is deterministic, and use a small N.
from backend.apps.agents.browser import browser_metrics as m
monkeypatch.setenv("OPENSWARM_BROWSER_METRICS_DIR", str(tmp_path))
m.P_METRICS_DIR_CACHE = None
m.P_TASK_COUNT = 0
monkeypatch.setattr(m, "P_AUDIT_EVERY_N", 5)
# autospec keeps the stub honest to threading.Thread's real signature (target/
# name/daemon); start() just runs the captured target synchronously.
thread = mocker.patch.object(m.threading, "Thread", autospec=True)
thread.return_value.start = mocker.Mock(
side_effect=lambda: thread.call_args.kwargs["target"]()
)
log = [{"tool": "BrowserClickIndex", "elapsed_ms": 5, "result_summary": "ok"}]
report = tmp_path / "self_audit_report.md"
for i in range(4):
m.record_task(f"s{i}", "b1", "t", "completed", 0, 7, log, {})
assert not report.exists(), "audit fired before N tasks"
m.record_task("s5", "b1", "t", "completed", 0, 7, log, {})
assert report.exists(), "audit did not fire at the Nth task"
def test_run_and_write_emits_a_report_file_and_never_raises():
d = tempfile.mkdtemp()
write_rows(d, "tasks.jsonl", [])
write_rows(d, "skill_events.jsonl", [])
path = audit.run_and_write(d)
assert path and os.path.exists(path)
# also safe on a totally missing dir
assert audit.run_and_write("/nonexistent/dir/xyz") in (None, os.path.join("/nonexistent/dir/xyz", "self_audit_report.md")) or True