From 50355206292875725aa228cbca3e044f98916193 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Tue, 4 Aug 2026 17:40:39 -0700 Subject: [PATCH] [eric] browser: prove a send by the posted content, not only by an emptied box --- .../agents/browser/browser_send_script.py | 20 +++++++++++++++++++ scripts/browser_canary.py | 8 ++++++-- 2 files changed, 26 insertions(+), 2 deletions(-) diff --git a/backend/apps/agents/browser/browser_send_script.py b/backend/apps/agents/browser/browser_send_script.py index 5622d7d6..83d77b41 100644 --- a/backend/apps/agents/browser/browser_send_script.py +++ b/backend/apps/agents/browser/browser_send_script.py @@ -148,6 +148,26 @@ async def complete_send( sent = True break p_why = f"payload-still-in-a-textbox (textbox rows={sum(1 for x in state3.splitlines() if ' # the post was really there (the cleanup that followed deleted it), the read just never # surfaced its text. Absence only counts as evidence once the read itself is known good, so # require proof we saw the destination at all before believing what we did not see on it. - saw_page = any(k in log for k in ("[browser-action] BrowserGetText", - "[browser-action] BrowserListInteractives")) + # Any evidence the audit run actually looked at a page. The first version of this listed two + # exact "[browser-action] X" strings that the log does not emit for reads, so `saw_page` was + # always False and every audit came back "unprovable" no matter what it saw. An instrument that + # can only ever return "don't know" is worse than none, because it looks like data. + saw_page = any(k in log for k in ("BrowserGetText", "BrowserListInteractives", + "dryrun-report", "browser-action", "browser-time")) hit = any(marker in line for line in log.splitlines() if "browser-action" not in line and "prompt" not in line.lower() and "canary-audit" not in line)