mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-10-01 22:14:51 +02:00
[eric] browser: refine ghost detector so read tasks need returned content, not state change
This commit is contained in:
@@ -0,0 +1,71 @@
|
||||
"""Ghost-success detector (scripts/analyze-browser-metrics.py): a 'completed'
|
||||
task with no verifiable work must be flagged, but honest reads/actions must not.
|
||||
This is the guard against features that fail silently without erroring."""
|
||||
|
||||
import importlib.util
|
||||
import os
|
||||
|
||||
_SCRIPT = os.path.join(os.path.dirname(__file__), "..", "..", "scripts", "analyze-browser-metrics.py")
|
||||
|
||||
|
||||
def _load():
|
||||
spec = importlib.util.spec_from_file_location("abm", _SCRIPT)
|
||||
m = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(m)
|
||||
return m
|
||||
|
||||
|
||||
def _ev(tool, ok=True, loop=False, result_len=50):
|
||||
return {"tool": tool, "ok": ok, "is_loop": loop, "result_len": result_len}
|
||||
|
||||
|
||||
def test_zero_tools_is_ghost():
|
||||
gv = _load().ghost_verdict
|
||||
is_ghost, reasons = gv({"status": "completed"}, [])
|
||||
assert is_ghost and any("ZERO" in r for r in reasons)
|
||||
|
||||
|
||||
def test_read_with_content_is_not_ghost():
|
||||
# a pure read task that actually returned data is legitimate work
|
||||
gv = _load().ghost_verdict
|
||||
is_ghost, _ = gv({"status": "completed"}, [_ev("BrowserGetText", result_len=120)])
|
||||
assert not is_ghost
|
||||
|
||||
|
||||
def test_empty_reads_only_is_ghost():
|
||||
gv = _load().ghost_verdict
|
||||
is_ghost, reasons = gv({"status": "completed"}, [
|
||||
_ev("BrowserGetText", result_len=0), _ev("BrowserScreenshot", result_len=0)])
|
||||
assert is_ghost
|
||||
|
||||
|
||||
def test_all_productive_actions_errored_is_ghost():
|
||||
gv = _load().ghost_verdict
|
||||
is_ghost, _ = gv({"status": "completed"}, [
|
||||
_ev("BrowserClickIndex", ok=False), _ev("BrowserClickIndex", ok=False)])
|
||||
assert is_ghost
|
||||
|
||||
|
||||
def test_honest_action_is_not_ghost():
|
||||
gv = _load().ghost_verdict
|
||||
is_ghost, _ = gv({"status": "completed"}, [_ev("BrowserNavigate"), _ev("BrowserClickIndex")])
|
||||
assert not is_ghost
|
||||
|
||||
|
||||
def test_loop_during_completed_is_ghost():
|
||||
gv = _load().ghost_verdict
|
||||
is_ghost, reasons = gv({"status": "completed"}, [_ev("BrowserNavigate"), _ev("BrowserClickIndex", loop=True)])
|
||||
assert is_ghost and any("loop" in r.lower() for r in reasons)
|
||||
|
||||
|
||||
def test_errored_task_is_never_ghost():
|
||||
# an honest failure (status=error) is not a ghost; ghosts are fake successes
|
||||
gv = _load().ghost_verdict
|
||||
assert not gv({"status": "error"}, [])[0]
|
||||
|
||||
|
||||
def test_tier_mapping_present():
|
||||
# the analyzer's _PRODUCTIVE set must include the real mutation tools
|
||||
m = _load()
|
||||
for t in ("BrowserClick", "BrowserClickIndex", "BrowserType", "BrowserNavigate", "BrowserReplayRoute"):
|
||||
assert t in m._PRODUCTIVE
|
||||
@@ -59,10 +59,19 @@ def ghost_verdict(task, events_for_task):
|
||||
productive = [t for t in tools if t in _PRODUCTIVE]
|
||||
errs = sum(1 for e in events_for_task if not e.get("ok"))
|
||||
total = len(events_for_task)
|
||||
# A read/extract task legitimately has no state-changing action; its evidence
|
||||
# is that a READ tool actually returned content. So "no productive action" is
|
||||
# only a ghost when NO read returned data either (i.e. nothing real happened).
|
||||
_READ = {"BrowserGetText", "BrowserGetElements", "BrowserListInteractives",
|
||||
"BrowserListRoutes", "BrowserReplayRoute", "BrowserScreenshot", "BrowserEvaluate"}
|
||||
read_with_content = any(
|
||||
e["tool"] in _READ and e.get("ok") and (e.get("result_len", 0) or 0) > 0
|
||||
for e in events_for_task
|
||||
)
|
||||
if total == 0:
|
||||
reasons.append("completed with ZERO tool calls (model declared done without acting)")
|
||||
if total and not productive:
|
||||
reasons.append("no state-changing action ran (only reads/meta) yet marked completed")
|
||||
if total and not productive and not read_with_content:
|
||||
reasons.append("no state-changing action AND no read returned content, yet marked completed")
|
||||
if total and errs / total >= 0.5:
|
||||
reasons.append(f"{errs}/{total} tool calls errored but still marked completed")
|
||||
if any(e.get("is_loop") for e in events_for_task):
|
||||
|
||||
Reference in New Issue
Block a user