From 69c45467351cf93b7ea59d89ec103282874d9139 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Tue, 2 Jun 2026 12:35:12 -0700 Subject: [PATCH] [eric] browser: refine ghost detector so read tasks need returned content, not state change --- backend/tests/test_ghost_detector.py | 71 ++++++++++++++++++++++++++++ scripts/analyze-browser-metrics.py | 13 ++++- 2 files changed, 82 insertions(+), 2 deletions(-) create mode 100644 backend/tests/test_ghost_detector.py diff --git a/backend/tests/test_ghost_detector.py b/backend/tests/test_ghost_detector.py new file mode 100644 index 00000000..d0660402 --- /dev/null +++ b/backend/tests/test_ghost_detector.py @@ -0,0 +1,71 @@ +"""Ghost-success detector (scripts/analyze-browser-metrics.py): a 'completed' +task with no verifiable work must be flagged, but honest reads/actions must not. +This is the guard against features that fail silently without erroring.""" + +import importlib.util +import os + +_SCRIPT = os.path.join(os.path.dirname(__file__), "..", "..", "scripts", "analyze-browser-metrics.py") + + +def _load(): + spec = importlib.util.spec_from_file_location("abm", _SCRIPT) + m = importlib.util.module_from_spec(spec) + spec.loader.exec_module(m) + return m + + +def _ev(tool, ok=True, loop=False, result_len=50): + return {"tool": tool, "ok": ok, "is_loop": loop, "result_len": result_len} + + +def test_zero_tools_is_ghost(): + gv = _load().ghost_verdict + is_ghost, reasons = gv({"status": "completed"}, []) + assert is_ghost and any("ZERO" in r for r in reasons) + + +def test_read_with_content_is_not_ghost(): + # a pure read task that actually returned data is legitimate work + gv = _load().ghost_verdict + is_ghost, _ = gv({"status": "completed"}, [_ev("BrowserGetText", result_len=120)]) + assert not is_ghost + + +def test_empty_reads_only_is_ghost(): + gv = _load().ghost_verdict + is_ghost, reasons = gv({"status": "completed"}, [ + _ev("BrowserGetText", result_len=0), _ev("BrowserScreenshot", result_len=0)]) + assert is_ghost + + +def test_all_productive_actions_errored_is_ghost(): + gv = _load().ghost_verdict + is_ghost, _ = gv({"status": "completed"}, [ + _ev("BrowserClickIndex", ok=False), _ev("BrowserClickIndex", ok=False)]) + assert is_ghost + + +def test_honest_action_is_not_ghost(): + gv = _load().ghost_verdict + is_ghost, _ = gv({"status": "completed"}, [_ev("BrowserNavigate"), _ev("BrowserClickIndex")]) + assert not is_ghost + + +def test_loop_during_completed_is_ghost(): + gv = _load().ghost_verdict + is_ghost, reasons = gv({"status": "completed"}, [_ev("BrowserNavigate"), _ev("BrowserClickIndex", loop=True)]) + assert is_ghost and any("loop" in r.lower() for r in reasons) + + +def test_errored_task_is_never_ghost(): + # an honest failure (status=error) is not a ghost; ghosts are fake successes + gv = _load().ghost_verdict + assert not gv({"status": "error"}, [])[0] + + +def test_tier_mapping_present(): + # the analyzer's _PRODUCTIVE set must include the real mutation tools + m = _load() + for t in ("BrowserClick", "BrowserClickIndex", "BrowserType", "BrowserNavigate", "BrowserReplayRoute"): + assert t in m._PRODUCTIVE diff --git a/scripts/analyze-browser-metrics.py b/scripts/analyze-browser-metrics.py index 27d38dab..631c5cd3 100755 --- a/scripts/analyze-browser-metrics.py +++ b/scripts/analyze-browser-metrics.py @@ -59,10 +59,19 @@ def ghost_verdict(task, events_for_task): productive = [t for t in tools if t in _PRODUCTIVE] errs = sum(1 for e in events_for_task if not e.get("ok")) total = len(events_for_task) + # A read/extract task legitimately has no state-changing action; its evidence + # is that a READ tool actually returned content. So "no productive action" is + # only a ghost when NO read returned data either (i.e. nothing real happened). + _READ = {"BrowserGetText", "BrowserGetElements", "BrowserListInteractives", + "BrowserListRoutes", "BrowserReplayRoute", "BrowserScreenshot", "BrowserEvaluate"} + read_with_content = any( + e["tool"] in _READ and e.get("ok") and (e.get("result_len", 0) or 0) > 0 + for e in events_for_task + ) if total == 0: reasons.append("completed with ZERO tool calls (model declared done without acting)") - if total and not productive: - reasons.append("no state-changing action ran (only reads/meta) yet marked completed") + if total and not productive and not read_with_content: + reasons.append("no state-changing action AND no read returned content, yet marked completed") if total and errs / total >= 0.5: reasons.append(f"{errs}/{total} tool calls errored but still marked completed") if any(e.get("is_loop") for e in events_for_task):