mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-08-22 20:52:23 +02:00
152 lines
5.9 KiB
Python
Executable File
152 lines
5.9 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
Analyze recorded browser-agent metrics and flag GHOST SUCCESS: tasks that
|
|
report "completed" but show no evidence of real work (errors-only, looped, or
|
|
zero productive mutation). A task ending without an exception is NOT proof the
|
|
goal was achieved, this is the independent reality check.
|
|
|
|
Usage:
|
|
python3 scripts/analyze-browser-metrics.py [metrics_dir]
|
|
Defaults to $OPENSWARM_BROWSER_METRICS_DIR, else
|
|
~/Library/Application Support/OpenSwarm/data/browser_metrics (mac) or backend/data/.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
from collections import Counter, defaultdict
|
|
|
|
# Tools that actually change page state (vs. read/meta). A "completed" task that
|
|
# never ran one of these did nothing but look around, suspicious.
|
|
_PRODUCTIVE = {
|
|
"BrowserClick", "BrowserClickIndex", "BrowserType", "BrowserNavigate",
|
|
"BrowserPressKey", "BrowserBatch", "BrowserReplayRoute",
|
|
}
|
|
|
|
|
|
def _default_dir():
|
|
env = os.environ.get("OPENSWARM_BROWSER_METRICS_DIR")
|
|
if env:
|
|
return env
|
|
mac = os.path.expanduser("~/Library/Application Support/OpenSwarm/data/browser_metrics")
|
|
if os.path.isdir(mac):
|
|
return mac
|
|
return os.path.join(os.path.dirname(__file__), "..", "backend", "data", "browser_metrics")
|
|
|
|
|
|
def _load(path):
|
|
if not os.path.exists(path):
|
|
return []
|
|
out = []
|
|
with open(path) as f:
|
|
for line in f:
|
|
line = line.strip()
|
|
if line:
|
|
try:
|
|
out.append(json.loads(line))
|
|
except Exception:
|
|
pass
|
|
return out
|
|
|
|
|
|
def ghost_verdict(task, events_for_task):
|
|
"""Return (is_ghost, [reasons]). Conservative: only flags 'completed' tasks
|
|
that show no evidence the work happened."""
|
|
if task.get("status") != "completed":
|
|
return False, []
|
|
reasons = []
|
|
tools = [e["tool"] for e in events_for_task]
|
|
productive = [t for t in tools if t in _PRODUCTIVE]
|
|
errs = sum(1 for e in events_for_task if not e.get("ok"))
|
|
total = len(events_for_task)
|
|
# A read/extract task legitimately has no state-changing action; its evidence
|
|
# is that a READ tool actually returned content. So "no productive action" is
|
|
# only a ghost when NO read returned data either (i.e. nothing real happened).
|
|
_READ = {"BrowserGetText", "BrowserGetElements", "BrowserListInteractives",
|
|
"BrowserListRoutes", "BrowserReplayRoute", "BrowserScreenshot", "BrowserEvaluate"}
|
|
read_with_content = any(
|
|
e["tool"] in _READ and e.get("ok") and (e.get("result_len", 0) or 0) > 0
|
|
for e in events_for_task
|
|
)
|
|
if total == 0:
|
|
reasons.append("completed with ZERO tool calls (model declared done without acting)")
|
|
if total and not productive and not read_with_content:
|
|
reasons.append("no state-changing action AND no read returned content, yet marked completed")
|
|
if total and errs / total >= 0.5:
|
|
reasons.append(f"{errs}/{total} tool calls errored but still marked completed")
|
|
if any(e.get("is_loop") for e in events_for_task):
|
|
reasons.append("loop detector fired during a 'completed' task")
|
|
prod_ok = [e for e in events_for_task if e["tool"] in _PRODUCTIVE and e.get("ok")]
|
|
if productive and not prod_ok:
|
|
reasons.append("every state-changing action errored, yet marked completed")
|
|
return (len(reasons) > 0), reasons
|
|
|
|
|
|
def main():
|
|
d = sys.argv[1] if len(sys.argv) > 1 else _default_dir()
|
|
events = _load(os.path.join(d, "events.jsonl"))
|
|
tasks = _load(os.path.join(d, "tasks.jsonl"))
|
|
print(f"metrics dir: {d}")
|
|
print(f"events: {len(events)} tasks: {len(tasks)}\n")
|
|
if not tasks and not events:
|
|
print("No metrics recorded yet. Run some browser-agent tasks first.")
|
|
return
|
|
|
|
ev_by_session = defaultdict(list)
|
|
for e in events:
|
|
ev_by_session[e.get("session_id")].append(e)
|
|
|
|
# Per-tier rollup across all events
|
|
tier_calls, tier_ms, tier_err = Counter(), Counter(), Counter()
|
|
for e in events:
|
|
t = e.get("tier", "other")
|
|
tier_calls[t] += 1
|
|
tier_ms[t] += int(e.get("elapsed_ms", 0) or 0)
|
|
if not e.get("ok"):
|
|
tier_err[t] += 1
|
|
|
|
print("=== PER-TIER (across all tool calls) ===")
|
|
print(f"{'tier':<20}{'calls':>6}{'avg_ms':>9}{'err%':>7}")
|
|
for t in sorted(tier_calls, key=lambda x: -tier_calls[x]):
|
|
c = tier_calls[t]
|
|
avg = round(tier_ms[t] / c, 1) if c else 0
|
|
errp = round(100 * tier_err[t] / c, 1) if c else 0
|
|
print(f"{t:<20}{c:>6}{avg:>9}{errp:>6}%")
|
|
|
|
print("\n=== PER-TASK (completion, time, cost, ghost check) ===")
|
|
completed = ghosts = 0
|
|
all_errs = Counter()
|
|
for tk in tasks:
|
|
evs = ev_by_session.get(tk.get("session_id"), [])
|
|
is_ghost, reasons = ghost_verdict(tk, evs)
|
|
if tk.get("status") == "completed":
|
|
completed += 1
|
|
if is_ghost:
|
|
ghosts += 1
|
|
for err, n in tk.get("recurring_errors", []):
|
|
all_errs[err] += n
|
|
flag = " ⚠️ GHOST" if is_ghost else ""
|
|
print(f"- [{tk.get('status')}] {tk.get('total_ms')}ms turns={tk.get('turns')} "
|
|
f"tools={tk.get('tool_calls')} tok_in={tk.get('tokens_in')} "
|
|
f"tok_out={tk.get('tokens_out')} :: {str(tk.get('task'))[:60]}{flag}")
|
|
for r in reasons:
|
|
print(f" ghost-reason: {r}")
|
|
|
|
print("\n=== RECURRING ERRORS (top 10 across tasks) ===")
|
|
for err, n in all_errs.most_common(10):
|
|
print(f" x{n} {err}")
|
|
|
|
print("\n=== SUMMARY ===")
|
|
n = len(tasks)
|
|
print(f"tasks: {n} completed: {completed} ghost-completed: {ghosts}")
|
|
if n:
|
|
avg_ms = round(sum(t.get("total_ms", 0) for t in tasks) / n)
|
|
avg_tok = round(sum(t.get("tokens_in", 0) + t.get("tokens_out", 0) for t in tasks) / n)
|
|
print(f"avg task time: {avg_ms}ms avg tokens/task: {avg_tok}")
|
|
print(f"honest completion rate: {round(100*(completed-ghosts)/n,1)}% "
|
|
f"(completed minus ghosts)")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|