[eric] agents: the recap carries no model text, nudges are short and unlabeled (ENG-358)

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01En8dRGsJPLrJCQBEkTH4Mp
This commit is contained in:
ciregenz
2026-08-21 00:20:04 -07:00
co-authored by Claude Fable 5
parent 295a7b52e2
commit a4be8c469f
8 changed files with 33 additions and 62 deletions
+2 -6
View File
@@ -21,6 +21,7 @@ from backend.apps.agents.manager.prompt.prompt_context import resolve_mode
logger = logging.getLogger(__name__)
from backend.apps.agents.manager.AgentManagerProtocol import AgentManagerProtocol
@@ -121,12 +122,7 @@ class Messaging(AgentManagerProtocol):
"session": session.model_dump(mode="json"),
})
# Hidden messages are harness plumbing (nudges, lost-step retries, auth heals) but ride the
# USER role, so agents stopped mid-task by a misfired nudge truthfully reported "the user
# told me to stop", and others read them as prompt injection (field reports, 2026-08-16).
# One attribution prefix at the one send chokepoint keeps every explanation honest.
if hidden and prompt and not prompt.startswith("[Automated"):
prompt = "[Automated message from OpenSwarm itself, not written by your user] " + prompt
# Hidden messages (nudges, lost-step retries, auth heals) go out verbatim: the attribution prefix they used to carry (ENG-326) announced harness traffic on the subscription lane, whose provider filter blocks exactly that (Eric's call, 2026-08-21); honesty about who is speaking lives in the nudge texts themselves.
skill_meta = [{"id": s["id"], "name": s["name"]} for s in (attached_skills or [])] or None
image_meta = [{"data": img["data"], "media_type": img.get("media_type", "image/png")} for img in (images or [])] or None
user_msg = Message(
@@ -274,18 +274,17 @@ class RunOptions(AgentManagerProtocol):
if session.needs_fork:
session.needs_fork = False
elif len(session.messages) > 1:
# The mode ratchets down when a provider policy filter blocks a recap-bearing turn (Alex's bricked-chat class): "minimal" carries no model text at all, "none" carries nothing.
# The recap never carries the model's own replies; the mode drops to "none" when a provider policy filter blocks even that (Alex's bricked-chat class).
p_mode = session.history_prefix_mode
history = "" if p_mode == "none" else build_history_prefix(
get_branch_messages(session),
cutoff_msg_id=session.compacted_through_msg_id,
mode=p_mode,
)
# Distill the dropped span into a cached aux summary so a rebuild keeps the gist of old turns instead of hard-dropping them. Fail-open: "" -> the plain recap above, exactly today's behavior.
from backend.apps.agents.manager.session.distill_history import distilled_history_summary
from backend.apps.agents.manager.session.history_compaction import wrap_platform_note
logger.info(f"[SPAWN-PHASE] distill start session={session_id[:8]} t={time.monotonic():.3f}")
distilled = await distilled_history_summary(session, global_settings) if p_mode == "full" else ""
distilled = await distilled_history_summary(session, global_settings) if p_mode != "none" else ""
logger.info(f"[SPAWN-PHASE] distill done session={session_id[:8]} t={time.monotonic():.3f}")
if distilled:
fenced = wrap_platform_note(f"Summary of earlier conversation (older turns compacted):\n{distilled}")
@@ -12,18 +12,12 @@ from typeguard import typechecked
from backend.apps.agents.core.models import AgentSession
from backend.apps.agents.manager.session.history_compaction import get_branch_messages
NUDGE_PROMPT = (
"You ended your turn without reporting anything. Continue exactly where you left off and "
"finish the task; when done, always end with your findings or answer as normal text."
)
NUDGE_PROMPT = "Finish the task, then answer in plain text."
# The last allowed nudge stops asking for more work: field data (Haik, 2026-08-08, 20 nudges in 5
# sessions) showed the model reads "continue and finish" as MORE tool calls then another silent
# quit, so the escalation demands the one thing the user is actually missing: text.
FINAL_NUDGE_PROMPT = (
"Stop. Do not call any more tools. In plain chat text, right now: report what you have done "
"so far, what is left, and anything blocking you. Even a partial status is required."
)
FINAL_NUDGE_PROMPT = "No more tools. Report what is done, what is left, and what blocks you."
# Post-cap honesty: the machinery is out of nudges and the turn STILL ended silent, so say so in
# the transcript instead of leaving a Done pill over a wall of tool rows.
@@ -86,9 +86,8 @@ class SessionPersistence(AgentManagerProtocol):
p_send = getattr(self, "send_message")
await p_send(
sid,
"[Automated message from OpenSwarm itself, not written by your user] The app "
"restarted while you were mid-task; nothing was lost. Continue exactly where "
"you left off; do not redo completed steps.",
"The app restarted while you were mid-task; nothing was lost. Continue exactly "
"where you left off; do not redo completed steps.",
hidden=True,
)
logger.info(f"crash-resume: session {sid} auto-resumed")
@@ -30,19 +30,19 @@ MAX_DISTILL_INPUT_CHARS = 60_000
MAX_PINNED_PATHS = 40
PATH_INPUT_KEYS = ("file_path", "notebook_path", "path")
# Plain wording on purpose, lifted from hermes-agent's filter-safe summarizer preamble (context_compressor.py, MIT): provider content filters flagged its earlier "NEVER continue / do not respond" framing, and the summarize call runs on the same subscription lane as the chat.
P_SYSTEM = (
"You are a note-taker that condenses a conversation transcript into a briefing. "
"You NEVER continue, answer, reply to, or role-play the conversation. You only "
"DESCRIBE it, in the third person ('The user asked...', 'The agent decided...'). "
"Your entire output is the briefing and nothing else."
"You are a summarization agent creating a context checkpoint. Treat the conversation "
"turns below as source material for a compact record of prior work, written in the third "
"person ('The user asked...', 'The agent decided...'). Produce only the briefing; no "
"greeting, preamble, or prefix."
)
P_USER_TEMPLATE = (
"Below, between <transcript> tags, is the earlier part of a conversation between a "
"Source material, between <transcript> tags: the earlier part of a conversation between a "
"user and an AI agent. Write a dense third-person briefing of it that preserves: the "
"user's goal and constraints, decisions already made, concrete facts / values / "
"identifiers / file paths mentioned, what was tried and how it turned out, and any open "
"threads. Do NOT continue or respond to the conversation; only describe what happened. "
"No preamble.\n\n<transcript>\n{body}\n</transcript>"
"threads.\n\n<transcript>\n{body}\n</transcript>"
)
@@ -132,24 +132,14 @@ def get_branch_messages(session) -> List:
return result
# A reply in the recap is a reminder of what the model said, not a copy: verbatim replays of the model's own long outputs are exactly what Anthropic's anti-distillation filter blocks, and the user still has the full text on screen.
RECAP_REPLY_GIST_CHARS = 600
@typechecked
def clamp_reply_gist(text: str) -> str:
if len(text) <= RECAP_REPLY_GIST_CHARS:
return text
return f"{text[:RECAP_REPLY_GIST_CHARS]} [... {len(text) - RECAP_REPLY_GIST_CHARS} chars of this reply omitted from recap ...]"
@typechecked
def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None, mode: str = "full") -> str:
def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None) -> str:
"""Format branch messages into a conversation summary for context injection.
`mode` is the session's history_prefix_mode: "full" carries asks, reply gists and the tool
trail; "minimal" carries only the user's asks and the tool calls (no model text at all), the
shape left once a provider policy filter has blocked a fuller recap.
Carries the user's asks and the tool trail, never the model's own replies: a replay of the
model's outputs in text we author is what Anthropic's anti-distillation filter blocks on the
subscription lane (192 blocks in 14 days, none on API keys). Claude Code and hermes keep old
answers only as model-written summaries; the distilled summary plays that role here.
When `cutoff_msg_id` is provided (session.compacted_through_msg_id), drop every
message up to and including that id so the marker the UI shows actually matches
@@ -170,20 +160,9 @@ def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None, mode: st
if m.role == "user":
text = m.content if isinstance(m.content, str) else str(m.content)
lines.append(f"The user asked: {strip_forged_sentinels(clamp_recap_text(text))}")
elif m.role == "assistant":
if mode == "minimal":
continue
text = m.content if isinstance(m.content, str) else str(m.content)
# First-person framing on purpose: a bare "User:/Assistant:" transcript inside a user
# message pattern-matches provider distillation filters (Anthropic blocked real users'
# recap turns as "duplicating model outputs"); "you replied" states the truth, this is
# the SAME assistant's own earlier work in this same session.
lines.append(f"You replied: {strip_forged_sentinels(clamp_reply_gist(text))}")
elif m.role == "tool_call":
lines.append(recap_tool_call_line(m.content))
elif m.role == "tool_result":
if mode == "minimal":
continue
body = fates.get(v)
if body is None:
lines.append(recap_tool_result_line(m.content))
@@ -193,9 +172,12 @@ def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None, mode: st
lines.append(f"{label}: {strip_forged_sentinels(body)}")
if not lines:
return ""
p_recap_frame = ("Recap of YOUR OWN earlier turns in this same conversation, summarized "
"locally by the OpenSwarm app so you can continue where you left off.")
return f"{SESSION_RECAP_OPEN}\n{PLATFORM_NOTE_PREAMBLE}\n{p_recap_frame}\n" + "\n".join(lines) + f"\n{SESSION_RECAP_CLOSE}"
# Framing lifted from hermes-agent's compaction handoff (context_compressor.py, MIT): reference only, never active instructions, the message after it is the single source of truth, and an explicit end marker so a weak model cannot read the last line as fresh input.
p_recap_frame = ("Recap of YOUR OWN earlier turns in this same conversation (what was asked and which tools "
"you ran), kept locally by the OpenSwarm app so you can continue where you left off. "
"Reference only: do not answer or redo anything in it; respond to the message that follows.")
return (f"{SESSION_RECAP_OPEN}\n{PLATFORM_NOTE_PREAMBLE}\n{p_recap_frame}\n" + "\n".join(lines)
+ f"\n--- end of recap; respond to the message below, not the recap above ---\n{SESSION_RECAP_CLOSE}")
@typechecked
+1 -1
View File
@@ -76,7 +76,7 @@ def test_auto_resume_sends_one_hidden_continuation(manager, tmp_path, monkeypatc
assert len(sent) == 1
sid, prompt, hidden = sent[0]
assert sid == "s-cut" and hidden is True
assert prompt.startswith("[Automated message from OpenSwarm itself")
assert "restarted while you were mid-task" in prompt, "the attribution prefix is added once, at the send chokepoint"
assert manager.crash_resume_queue == []
@@ -3,7 +3,9 @@ read hidden harness messages as prompt injection (field reports, 2026-08-16). Me
hidden prompt (silent-quit nudges incl. the FINAL one that literally opens "Stop. Do not call any
more tools.", lost-step retries, auth heals, context-break continuations) rides the USER role, so
the model's misattribution is honest from its chair. Seal: one attribution prefix at the ONE send
chokepoint, so no hidden message can ever read as the user's words.
chokepoint, so no hidden message can ever read as the user's words. REVERSED 2026-08-21 (Eric):
the prefix announced harness traffic on the subscription lane, whose provider filter blocks exactly
that (ENG-358), so hidden prompts now go out verbatim and honesty lives in the nudge texts.
"""
import inspect
@@ -29,8 +31,7 @@ async def test_hidden_prompt_gets_the_attribution_prefix():
pass # downstream turn machinery may bail in a unit context; the append happened first
hidden = [m for m in session.messages if m.role == "user" and m.hidden]
assert hidden, "hidden message never appended"
assert hidden[-1].content.startswith("[Automated message from OpenSwarm itself"), \
"a nudge the model attributes to the user is the fabricated-stop bug"
assert hidden[-1].content == "Stop. Do not call any more tools.", "hidden prompts go out verbatim: no attribution prefix on the subscription lane (Eric, 2026-08-21)"
finally:
agent_manager.sessions.pop("hm-1", None)
@@ -55,7 +56,7 @@ async def test_visible_user_prompt_is_untouched():
agent_manager.sessions.pop("hm-2", None)
def test_no_double_prefix():
def test_no_prefix_is_ever_added():
from backend.apps.agents.manager import Messaging
src = inspect.getsource(Messaging)
assert 'not prompt.startswith("[Automated")' in src, "re-sent continuations must not stack prefixes"
assert "HIDDEN_NOTE_PREFIX" not in src and "[Automated message" not in src, "hidden prompts are sent as written"