mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-28 12:34:50 +02:00
[eric] browser: widen secret redaction (2fa codes, credential fields), 0600 metrics files, scrub task prompts
This commit is contained in:
@@ -21,6 +21,7 @@ Files (under DATA_ROOT/browser_metrics/, env-overridable):
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from collections import Counter
|
||||
|
||||
@@ -74,7 +75,7 @@ def _metrics_dir() -> str:
|
||||
import tempfile
|
||||
base = os.path.join(tempfile.gettempdir(), "openswarm_browser_metrics")
|
||||
try:
|
||||
os.makedirs(base, exist_ok=True)
|
||||
os.makedirs(base, mode=0o700, exist_ok=True)
|
||||
except Exception:
|
||||
pass
|
||||
_metrics_dir_cache = base
|
||||
@@ -84,12 +85,29 @@ def _metrics_dir() -> str:
|
||||
def _append(filename: str, obj: dict) -> None:
|
||||
try:
|
||||
path = os.path.join(_metrics_dir(), filename)
|
||||
with open(path, "a", encoding="utf-8") as f:
|
||||
# owner-only: these lines can carry task text and error snippets
|
||||
fd = os.open(path, os.O_APPEND | os.O_CREAT | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fd, "a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(obj, default=str) + "\n")
|
||||
except Exception as e:
|
||||
logger.debug(f"[browser-metrics] write failed: {e}")
|
||||
|
||||
|
||||
# A task prompt can carry a literal secret ("log in with password hunter2");
|
||||
# scrub the value before it lands in tasks.jsonl. Keyword+value and known
|
||||
# token prefixes only; the task's normal words stay greppable.
|
||||
_TASK_SECRET_RE = re.compile(
|
||||
r"\b(password|passcode|passphrase|pin|otp|token|secret|api[_-]?key)\b\s*(?:is|[:=])?\s*\S+",
|
||||
re.I,
|
||||
)
|
||||
_TASK_TOKEN_RE = re.compile(r"\b(sk-|ghp_|gho_|pk_|xox[bap]-|AIza|eyJ)[A-Za-z0-9_\-.]{8,}")
|
||||
|
||||
|
||||
def _scrub_task(task: str) -> str:
|
||||
t = _TASK_SECRET_RE.sub(lambda m: f"{m.group(1)} [redacted]", task or "")
|
||||
return _TASK_TOKEN_RE.sub("[redacted]", t)
|
||||
|
||||
|
||||
def record_tool(session_id, browser_id, turn, tool, elapsed_ms, ok, error,
|
||||
is_loop, stagnation_streak, result_len) -> None:
|
||||
"""One line per executed tool call. Best-effort."""
|
||||
@@ -153,7 +171,7 @@ def record_task(session_id, browser_id, task, status, started_at, turns,
|
||||
"ts": time.time(),
|
||||
"session_id": session_id,
|
||||
"browser_id": browser_id,
|
||||
"task": (task or "")[:200],
|
||||
"task": _scrub_task(task)[:200],
|
||||
"task_sig": task_sig,
|
||||
"path": path,
|
||||
"playbook_seeded": bool(playbook_seeded),
|
||||
|
||||
@@ -83,7 +83,7 @@ def _dir() -> str | None:
|
||||
except Exception:
|
||||
return None
|
||||
try:
|
||||
os.makedirs(base, exist_ok=True)
|
||||
os.makedirs(base, mode=0o700, exist_ok=True)
|
||||
except Exception:
|
||||
return None
|
||||
return base
|
||||
|
||||
@@ -100,13 +100,19 @@ _SSN_RE = re.compile(r"\b\d{3}-\d{2}-\d{4}\b")
|
||||
_CARD_RE = re.compile(r"\b(?:\d[ -]?){13,19}\b")
|
||||
_PHONE_RE = re.compile(r"\b(?:\+?\d[ -]?){10,15}\b")
|
||||
_TOKEN_PREFIX_RE = re.compile(r"\b(sk-|ghp_|gho_|pk_|xox[bap]-|AIza|eyJ)")
|
||||
_SENSITIVE_FIELD_RE = re.compile(r"pass|pwd|secret|otp|cvv|cvc|ssn|card|token|api[_-]?key|security", re.I)
|
||||
_SENSITIVE_FIELD_RE = re.compile(
|
||||
r"pass|pwd|secret|otp|cvv|cvc|ssn|card|token|api[_-]?key|security"
|
||||
r"|user|login|sign[-_]?in|email|auth|seed|recovery|phrase|\bpin\b|2fa|verif|code",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _looks_sensitive(text: str, selector: str = "") -> bool:
|
||||
"""Conservative: err toward 'sensitive' so secrets never persist. Catches
|
||||
emails, SSNs, card/phone-shaped digit runs, known key prefixes, long
|
||||
high-entropy tokens, and anything typed into a password-shaped field."""
|
||||
high-entropy tokens, bare one-time-code digit runs, and anything typed into
|
||||
a credential-shaped field (a wrongly-blocked persist just keeps the skill
|
||||
in-memory, so false positives are cheap; a leak is not)."""
|
||||
if selector and _SENSITIVE_FIELD_RE.search(selector):
|
||||
return True
|
||||
if not text:
|
||||
@@ -117,8 +123,11 @@ def _looks_sensitive(text: str, selector: str = "") -> bool:
|
||||
return True
|
||||
if _PHONE_RE.search(text):
|
||||
return True
|
||||
# long high-entropy token: >=20 chars with both letters and digits
|
||||
stripped = text.strip()
|
||||
# bare 6-8 digit run: the shape of every 2FA/SMS code; never worth persisting
|
||||
if re.fullmatch(r"\d{6,8}", stripped):
|
||||
return True
|
||||
# long high-entropy token: >=20 chars with both letters and digits
|
||||
if len(stripped) >= 20 and any(c.isdigit() for c in stripped) and any(c.isalpha() for c in stripped) and " " not in stripped:
|
||||
return True
|
||||
return False
|
||||
@@ -373,7 +382,7 @@ def _skills_dir() -> str | None:
|
||||
except Exception:
|
||||
return None
|
||||
try:
|
||||
os.makedirs(base, exist_ok=True)
|
||||
os.makedirs(base, mode=0o700, exist_ok=True)
|
||||
except Exception:
|
||||
return None
|
||||
return base
|
||||
|
||||
@@ -87,3 +87,19 @@ def test_metrics_never_raises_on_bad_dir(monkeypatch):
|
||||
bm.record_tool("s", "b", 1, "BrowserScreenshot", 5, ok=True, error="",
|
||||
is_loop=False, stagnation_streak=0, result_len=1) # must not raise
|
||||
bm.record_task("s", "b", "t", "error", __import__("time").time(), 1, [], {})
|
||||
|
||||
|
||||
def test_task_secrets_are_scrubbed_from_tasks_jsonl(tmp_path, monkeypatch):
|
||||
from backend.apps.agents.browser import browser_metrics as bm
|
||||
import os as _os
|
||||
import time as _time
|
||||
monkeypatch.setenv("OPENSWARM_BROWSER_METRICS_DIR", str(tmp_path))
|
||||
bm._metrics_dir_cache = None
|
||||
bm.record_task("s1", "b1", "log into acme with password hunter2 then post sk-abc12345678901234567",
|
||||
"completed", _time.time() - 1, 2, [], {})
|
||||
line = open(_os.path.join(str(tmp_path), "tasks.jsonl")).read()
|
||||
assert "hunter2" not in line and "sk-abc" not in line
|
||||
assert "password [redacted]" in line
|
||||
# owner-only file perms
|
||||
mode = _os.stat(_os.path.join(str(tmp_path), "tasks.jsonl")).st_mode & 0o777
|
||||
assert mode == 0o600
|
||||
|
||||
@@ -483,3 +483,16 @@ def test_extract_first_json_strips_fences_and_prose():
|
||||
assert _first_json('Here you go: [{"n": "x"}] hope that helps') == '[{"n": "x"}]'
|
||||
assert _first_json("no json here") == ""
|
||||
assert _first_json('{"broken": ') == ""
|
||||
|
||||
|
||||
def test_widened_redaction_catches_audit_bypasses():
|
||||
# the audit's three named bypasses: bare 2FA digits, credential-shaped
|
||||
# fields the old regex missed, and seed/recovery phrase boxes
|
||||
assert sk._looks_sensitive("481922", "")
|
||||
assert sk._looks_sensitive("hunter2", "#user")
|
||||
assert sk._looks_sensitive("me@corp.com", "#login-email")
|
||||
assert sk._looks_sensitive("correct horse battery staple", "#seed-phrase")
|
||||
assert sk._looks_sensitive("123456", "input[name='verification-code']")
|
||||
# the bread-and-butter skill (a search query) still persists
|
||||
assert not sk._looks_sensitive("shoes", "#search-input")
|
||||
assert not sk._looks_sensitive("Ada Lovelace", ".search-global-typeahead input")
|
||||
|
||||
Reference in New Issue
Block a user