[eric] agents: recap ages history hermes-style instead of dropping it; stubs keep the re-runnable command, duplicates collapse, newest results survive verbatim (ENG-354 endgame)

This commit is contained in:
ciregenz
2026-08-19 13:40:47 -07:00
parent 39d8da2697
commit 6a1dbc4212
3 changed files with 222 additions and 7 deletions
@@ -0,0 +1,115 @@
"""Hermes-style graded aging for the session recap (ENG-354 endgame; lifted from
NousResearch/hermes-agent agent/context_compressor.py::_prune_old_tool_results, MIT).
The old recap blind-head-truncated EVERY tool result at 500 chars (deleting exactly the
answers a long run needs back after a context break) and hard-dropped everything before the
compaction cutoff. Aging is recoverable instead of destructive: the newest results within a
token budget survive verbatim, small results survive whole, exact duplicates collapse to a
back-reference, and everything older becomes a one-line stub that KEEPS the tool name and
arguments, so the agent can re-run any call whose detail it still needs."""
import hashlib
import json
from typing import List, Optional, Tuple
from typeguard import typechecked
# Hermes floors: below this a result costs less than a summary of it would.
PRUNE_MIN_CHARS = 200
# Verbatim tail budget in chars (~3K tokens) plus a hard count floor, hermes-shaped.
TAIL_BUDGET_CHARS = 12_000
TAIL_COUNT_FLOOR = 7
# One verbatim survivor never eats the whole tail budget: middle-elide past this.
TAIL_ITEM_CAP = 6_000
STUB_ARGS_CAP = 160
DUPLICATE_LINE = "[Duplicate tool output — same content as a more recent call]"
@typechecked
def p_pair_calls(messages: List) -> dict:
"""index of each tool_result -> (tool_name, compact_args) from its nearest preceding call."""
pairs = {}
last_call: Tuple[str, str] = ("tool", "")
for i, m in enumerate(messages):
role = getattr(m, "role", "")
c = getattr(m, "content", None)
if role == "tool_call" and isinstance(c, dict):
tool = str(c.get("tool") or c.get("name") or "tool")
try:
args = json.dumps(c.get("input"), ensure_ascii=False, default=str)
except Exception:
args = str(c.get("input"))
last_call = (tool, args)
elif role == "tool_result":
name = c.get("tool_name") if isinstance(c, dict) else None
pairs[i] = (str(name) if name else last_call[0], last_call[1])
return pairs
@typechecked
def p_result_text(content: object) -> str:
if isinstance(content, dict):
text = content.get("text")
if isinstance(text, str):
return text
try:
return json.dumps(content, ensure_ascii=False, default=str)
except Exception:
return str(content)
return str(content)
@typechecked
def stub_line(tool: str, args: str, size: int) -> str:
"""The aged one-liner; the args survive so the call is re-runnable (the hermes property)."""
compact = args.strip()
if len(compact) > STUB_ARGS_CAP:
compact = compact[:STUB_ARGS_CAP] + "..."
return f"[{tool}] {compact} ({size:,} chars result)"
@typechecked
def elide_middle(text: str, cap: int) -> str:
if len(text) <= cap:
return text
head = int(cap * 0.6)
tail = cap - head
return f"{text[:head]}\n[... {len(text) - cap:,} chars elided ...]\n{text[-tail:]}"
@typechecked
def age_tool_results(messages: List, cutoff_idx: int = -1) -> dict:
"""Decide each tool_result's recap fate. Returns index -> recap body string.
Walking newest-first (hermes pass order): duplicates collapse to a back-reference,
the newest results within TAIL_BUDGET_CHARS (or the TAIL_COUNT_FLOOR, whichever
protects more) stay verbatim, small results always stay whole, and everything else,
plus everything at or before ``cutoff_idx``, ages into a stub."""
pairs = p_pair_calls(messages)
fates: dict = {}
seen_hashes: set = set()
tail_spent = 0
tail_kept = 0
for i in range(len(messages) - 1, -1, -1):
if i not in pairs:
continue
text = p_result_text(getattr(messages[i], "content", None))
tool, args = pairs[i]
if len(text) >= PRUNE_MIN_CHARS:
h = hashlib.md5(text.encode("utf-8", errors="replace")).hexdigest()[:12]
if h in seen_hashes:
fates[i] = DUPLICATE_LINE
continue
seen_hashes.add(h)
if len(text) < PRUNE_MIN_CHARS:
fates[i] = text
continue
in_tail = i > cutoff_idx and (tail_kept < TAIL_COUNT_FLOOR or tail_spent < TAIL_BUDGET_CHARS)
if in_tail:
kept = elide_middle(text, TAIL_ITEM_CAP)
fates[i] = kept
tail_spent += len(kept)
tail_kept += 1
else:
fates[i] = stub_line(tool, args, len(text))
return fates
@@ -140,14 +140,18 @@ def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None) -> str:
message up to and including that id so the marker the UI shows actually matches
what the model sees. Missing cutoff id falls through to full history.
"""
# Aging replaced dropping (ENG-354, hermes lift): pre-cutoff history becomes re-runnable
# one-line stubs instead of vanishing, duplicates collapse, and the newest tool results
# survive verbatim inside a budget, so a context break costs detail, never the trail.
from backend.apps.agents.manager.session.aged_recap_lines import age_tool_results
cutoff_idx = -1
if cutoff_msg_id:
skip_idx = next((i for i, m in enumerate(messages) if m.id == cutoff_msg_id), -1)
if skip_idx >= 0:
messages = messages[skip_idx + 1:]
cutoff_idx = next((i for i, m in enumerate(messages) if m.id == cutoff_msg_id), -1)
visible = [(i, m) for i, m in enumerate(messages) if not getattr(m, "hidden", False)]
fates = age_tool_results([m for _, m in visible], cutoff_idx=next(
(v for v, (i, _) in enumerate(visible) if i == cutoff_idx), -1))
lines = []
for m in messages:
if getattr(m, "hidden", False):
continue
for v, (i, m) in enumerate(visible):
if m.role == "user":
text = m.content if isinstance(m.content, str) else str(m.content)
lines.append(f"User: {strip_forged_sentinels(clamp_recap_text(text))}")
@@ -157,7 +161,13 @@ def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None) -> str:
elif m.role == "tool_call":
lines.append(recap_tool_call_line(m.content))
elif m.role == "tool_result":
lines.append(recap_tool_result_line(m.content))
body = fates.get(v)
if body is None:
lines.append(recap_tool_result_line(m.content))
else:
tool_name = m.content.get("tool_name") if isinstance(m.content, dict) else None
label = f"Tool result ({tool_name})" if tool_name else "Tool result"
lines.append(f"{label}: {strip_forged_sentinels(body)}")
if not lines:
return ""
return f"{SESSION_RECAP_OPEN}\n{PLATFORM_NOTE_PREAMBLE}\n" + "\n".join(lines) + f"\n{SESSION_RECAP_CLOSE}"
+90
View File
@@ -0,0 +1,90 @@
"""Pins the hermes-aging recap contract (ENG-354 endgame): duplicates collapse, the newest
results survive verbatim, old bulk becomes a re-runnable stub, pre-cutoff history ages
instead of vanishing, and the whole recap is bounded no matter how long the session ran."""
from backend.apps.agents.core.models import AgentSession, Message
from backend.apps.agents.manager.session.aged_recap_lines import (
DUPLICATE_LINE,
PRUNE_MIN_CHARS,
TAIL_BUDGET_CHARS,
age_tool_results,
stub_line,
)
from backend.apps.agents.manager.session.history_compaction import build_history_prefix
def p_msgs(*pairs):
out = []
for tool, args, result in pairs:
out.append(Message(role="tool_call", content={"tool": tool, "input": args}))
out.append(Message(role="tool_result", content={"text": result, "tool_name": tool}))
return out
def test_old_bulk_ages_into_a_rerunnable_stub():
msgs = p_msgs(*[("Read", {"file_path": f"/tmp/f{i}.txt"}, f"body{i} " + "x" * 5000) for i in range(12)])
fates = age_tool_results(msgs)
aged = [f for f in fates.values() if f.startswith("[Read]")]
assert aged, "no stubs produced"
assert any("/tmp/f0.txt" in f and "chars result)" in f for f in aged), "stub lost the re-runnable args"
def test_newest_results_survive_verbatim_within_budget():
msgs = p_msgs(*[("Read", {"file_path": f"/f{i}"}, f"UNIQUE-{i} " + "y" * 3000) for i in range(10)])
fates = age_tool_results(msgs)
newest = fates[len(msgs) - 1]
assert "UNIQUE-9" in newest and not newest.startswith("[Read]")
total_verbatim = sum(len(f) for f in fates.values() if not f.startswith("["))
# The count floor is hermes semantics: at least TAIL_COUNT_FLOOR survive regardless of budget, each capped at TAIL_ITEM_CAP.
from backend.apps.agents.manager.session.aged_recap_lines import TAIL_COUNT_FLOOR, TAIL_ITEM_CAP
assert total_verbatim <= max(TAIL_BUDGET_CHARS + TAIL_ITEM_CAP, TAIL_COUNT_FLOOR * TAIL_ITEM_CAP)
assert len([f for f in fates.values() if not f.startswith("[")]) >= TAIL_COUNT_FLOOR
def test_duplicates_collapse_to_backreference():
same = "identical result " + "z" * 400
msgs = p_msgs(("Read", {"file_path": "/a"}, same), ("Read", {"file_path": "/a"}, same), ("Read", {"file_path": "/a"}, same))
fates = age_tool_results(msgs)
dupes = [f for f in fates.values() if f == DUPLICATE_LINE]
assert len(dupes) == 2, "older duplicates must collapse; newest keeps the copy"
def test_small_results_always_survive_whole():
msgs = p_msgs(*[("Bash", {"command": f"echo {i}"}, f"tiny-{i}") for i in range(30)])
fates = age_tool_results(msgs)
assert all(f.startswith("tiny-") for f in fates.values())
def test_precutoff_ages_instead_of_vanishing():
s = AgentSession(name="t", model="sonnet")
for i in range(10):
s.messages.append(Message(role="tool_call", content={"tool": "Bash", "input": {"command": f"grep -rn pat{i} src/"}}))
s.messages.append(Message(role="tool_result", content={"text": f"match{i} " + "m" * 3000, "tool_name": "Bash"}))
cutoff = s.messages[9].id
recap = build_history_prefix(s.messages, cutoff_msg_id=cutoff)
assert "grep -rn pat1" in recap, "pre-cutoff command vanished; aging must keep the trail"
assert "match9" in recap, "newest result must survive verbatim"
pre = recap.split("match9")[0]
assert "chars result)" in pre, "pre-cutoff results must be stubs, not verbatim"
def test_recap_is_bounded_for_marathon_sessions():
s = AgentSession(name="t", model="sonnet")
for i in range(120):
s.messages.append(Message(role="tool_call", content={"tool": "Read", "input": {"file_path": f"/tmp/m{i}.txt"}}))
s.messages.append(Message(role="tool_result", content={"text": f"MARA-{i} " + "w" * 8000, "tool_name": "Read"}))
recap = build_history_prefix(s.messages)
assert len(recap) < 80_000, f"recap unbounded: {len(recap)} chars for a 120-read session"
assert "MARA-119" in recap
assert "/tmp/m0.txt" in recap
def test_stub_format_matches_hermes_shape():
line = stub_line("Bash", '{"command": "pytest -q"}', 22_770)
assert line == '[Bash] {"command": "pytest -q"} (22,770 chars result)'
def test_below_floor_never_stubbed():
msgs = p_msgs(("Bash", {"command": "date"}, "x" * (PRUNE_MIN_CHARS - 1)))
fates = age_tool_results(msgs)
assert not list(fates.values())[0].startswith("[Bash]")