[eric] browser: refine ghost detector so read tasks need returned content, not state change

This commit is contained in:
ciregenz
2026-06-02 12:35:12 -07:00
parent 8a0cb4a6a0
commit 69c4546735
2 changed files with 82 additions and 2 deletions
+71
View File
@@ -0,0 +1,71 @@
"""Ghost-success detector (scripts/analyze-browser-metrics.py): a 'completed'
task with no verifiable work must be flagged, but honest reads/actions must not.
This is the guard against features that fail silently without erroring."""
import importlib.util
import os
_SCRIPT = os.path.join(os.path.dirname(__file__), "..", "..", "scripts", "analyze-browser-metrics.py")
def _load():
spec = importlib.util.spec_from_file_location("abm", _SCRIPT)
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)
return m
def _ev(tool, ok=True, loop=False, result_len=50):
return {"tool": tool, "ok": ok, "is_loop": loop, "result_len": result_len}
def test_zero_tools_is_ghost():
gv = _load().ghost_verdict
is_ghost, reasons = gv({"status": "completed"}, [])
assert is_ghost and any("ZERO" in r for r in reasons)
def test_read_with_content_is_not_ghost():
# a pure read task that actually returned data is legitimate work
gv = _load().ghost_verdict
is_ghost, _ = gv({"status": "completed"}, [_ev("BrowserGetText", result_len=120)])
assert not is_ghost
def test_empty_reads_only_is_ghost():
gv = _load().ghost_verdict
is_ghost, reasons = gv({"status": "completed"}, [
_ev("BrowserGetText", result_len=0), _ev("BrowserScreenshot", result_len=0)])
assert is_ghost
def test_all_productive_actions_errored_is_ghost():
gv = _load().ghost_verdict
is_ghost, _ = gv({"status": "completed"}, [
_ev("BrowserClickIndex", ok=False), _ev("BrowserClickIndex", ok=False)])
assert is_ghost
def test_honest_action_is_not_ghost():
gv = _load().ghost_verdict
is_ghost, _ = gv({"status": "completed"}, [_ev("BrowserNavigate"), _ev("BrowserClickIndex")])
assert not is_ghost
def test_loop_during_completed_is_ghost():
gv = _load().ghost_verdict
is_ghost, reasons = gv({"status": "completed"}, [_ev("BrowserNavigate"), _ev("BrowserClickIndex", loop=True)])
assert is_ghost and any("loop" in r.lower() for r in reasons)
def test_errored_task_is_never_ghost():
# an honest failure (status=error) is not a ghost; ghosts are fake successes
gv = _load().ghost_verdict
assert not gv({"status": "error"}, [])[0]
def test_tier_mapping_present():
# the analyzer's _PRODUCTIVE set must include the real mutation tools
m = _load()
for t in ("BrowserClick", "BrowserClickIndex", "BrowserType", "BrowserNavigate", "BrowserReplayRoute"):
assert t in m._PRODUCTIVE
+11 -2
View File
@@ -59,10 +59,19 @@ def ghost_verdict(task, events_for_task):
productive = [t for t in tools if t in _PRODUCTIVE]
errs = sum(1 for e in events_for_task if not e.get("ok"))
total = len(events_for_task)
# A read/extract task legitimately has no state-changing action; its evidence
# is that a READ tool actually returned content. So "no productive action" is
# only a ghost when NO read returned data either (i.e. nothing real happened).
_READ = {"BrowserGetText", "BrowserGetElements", "BrowserListInteractives",
"BrowserListRoutes", "BrowserReplayRoute", "BrowserScreenshot", "BrowserEvaluate"}
read_with_content = any(
e["tool"] in _READ and e.get("ok") and (e.get("result_len", 0) or 0) > 0
for e in events_for_task
)
if total == 0:
reasons.append("completed with ZERO tool calls (model declared done without acting)")
if total and not productive:
reasons.append("no state-changing action ran (only reads/meta) yet marked completed")
if total and not productive and not read_with_content:
reasons.append("no state-changing action AND no read returned content, yet marked completed")
if total and errs / total >= 0.5:
reasons.append(f"{errs}/{total} tool calls errored but still marked completed")
if any(e.get("is_loop") for e in events_for_task):