diff --git a/backend/apps/agents/manager/session/aged_recap_lines.py b/backend/apps/agents/manager/session/aged_recap_lines.py new file mode 100644 index 00000000..e8c7bbfc --- /dev/null +++ b/backend/apps/agents/manager/session/aged_recap_lines.py @@ -0,0 +1,115 @@ +"""Hermes-style graded aging for the session recap (ENG-354 endgame; lifted from +NousResearch/hermes-agent agent/context_compressor.py::_prune_old_tool_results, MIT). + +The old recap blind-head-truncated EVERY tool result at 500 chars (deleting exactly the +answers a long run needs back after a context break) and hard-dropped everything before the +compaction cutoff. Aging is recoverable instead of destructive: the newest results within a +token budget survive verbatim, small results survive whole, exact duplicates collapse to a +back-reference, and everything older becomes a one-line stub that KEEPS the tool name and +arguments, so the agent can re-run any call whose detail it still needs.""" + +import hashlib +import json +from typing import List, Optional, Tuple + +from typeguard import typechecked + +# Hermes floors: below this a result costs less than a summary of it would. +PRUNE_MIN_CHARS = 200 +# Verbatim tail budget in chars (~3K tokens) plus a hard count floor, hermes-shaped. +TAIL_BUDGET_CHARS = 12_000 +TAIL_COUNT_FLOOR = 7 +# One verbatim survivor never eats the whole tail budget: middle-elide past this. +TAIL_ITEM_CAP = 6_000 +STUB_ARGS_CAP = 160 +DUPLICATE_LINE = "[Duplicate tool output — same content as a more recent call]" + + +@typechecked +def p_pair_calls(messages: List) -> dict: + """index of each tool_result -> (tool_name, compact_args) from its nearest preceding call.""" + pairs = {} + last_call: Tuple[str, str] = ("tool", "") + for i, m in enumerate(messages): + role = getattr(m, "role", "") + c = getattr(m, "content", None) + if role == "tool_call" and isinstance(c, dict): + tool = str(c.get("tool") or c.get("name") or "tool") + try: + args = json.dumps(c.get("input"), ensure_ascii=False, default=str) + except Exception: + args = str(c.get("input")) + last_call = (tool, args) + elif role == "tool_result": + name = c.get("tool_name") if isinstance(c, dict) else None + pairs[i] = (str(name) if name else last_call[0], last_call[1]) + return pairs + + +@typechecked +def p_result_text(content: object) -> str: + if isinstance(content, dict): + text = content.get("text") + if isinstance(text, str): + return text + try: + return json.dumps(content, ensure_ascii=False, default=str) + except Exception: + return str(content) + return str(content) + + +@typechecked +def stub_line(tool: str, args: str, size: int) -> str: + """The aged one-liner; the args survive so the call is re-runnable (the hermes property).""" + compact = args.strip() + if len(compact) > STUB_ARGS_CAP: + compact = compact[:STUB_ARGS_CAP] + "..." + return f"[{tool}] {compact} ({size:,} chars result)" + + +@typechecked +def elide_middle(text: str, cap: int) -> str: + if len(text) <= cap: + return text + head = int(cap * 0.6) + tail = cap - head + return f"{text[:head]}\n[... {len(text) - cap:,} chars elided ...]\n{text[-tail:]}" + + +@typechecked +def age_tool_results(messages: List, cutoff_idx: int = -1) -> dict: + """Decide each tool_result's recap fate. Returns index -> recap body string. + + Walking newest-first (hermes pass order): duplicates collapse to a back-reference, + the newest results within TAIL_BUDGET_CHARS (or the TAIL_COUNT_FLOOR, whichever + protects more) stay verbatim, small results always stay whole, and everything else, + plus everything at or before ``cutoff_idx``, ages into a stub.""" + pairs = p_pair_calls(messages) + fates: dict = {} + seen_hashes: set = set() + tail_spent = 0 + tail_kept = 0 + for i in range(len(messages) - 1, -1, -1): + if i not in pairs: + continue + text = p_result_text(getattr(messages[i], "content", None)) + tool, args = pairs[i] + if len(text) >= PRUNE_MIN_CHARS: + h = hashlib.md5(text.encode("utf-8", errors="replace")).hexdigest()[:12] + if h in seen_hashes: + fates[i] = DUPLICATE_LINE + continue + seen_hashes.add(h) + if len(text) < PRUNE_MIN_CHARS: + fates[i] = text + continue + in_tail = i > cutoff_idx and (tail_kept < TAIL_COUNT_FLOOR or tail_spent < TAIL_BUDGET_CHARS) + if in_tail: + kept = elide_middle(text, TAIL_ITEM_CAP) + fates[i] = kept + tail_spent += len(kept) + tail_kept += 1 + else: + fates[i] = stub_line(tool, args, len(text)) + return fates diff --git a/backend/apps/agents/manager/session/history_compaction.py b/backend/apps/agents/manager/session/history_compaction.py index cc778354..733b2bc0 100644 --- a/backend/apps/agents/manager/session/history_compaction.py +++ b/backend/apps/agents/manager/session/history_compaction.py @@ -140,14 +140,18 @@ def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None) -> str: message up to and including that id so the marker the UI shows actually matches what the model sees. Missing cutoff id falls through to full history. """ + # Aging replaced dropping (ENG-354, hermes lift): pre-cutoff history becomes re-runnable + # one-line stubs instead of vanishing, duplicates collapse, and the newest tool results + # survive verbatim inside a budget, so a context break costs detail, never the trail. + from backend.apps.agents.manager.session.aged_recap_lines import age_tool_results + cutoff_idx = -1 if cutoff_msg_id: - skip_idx = next((i for i, m in enumerate(messages) if m.id == cutoff_msg_id), -1) - if skip_idx >= 0: - messages = messages[skip_idx + 1:] + cutoff_idx = next((i for i, m in enumerate(messages) if m.id == cutoff_msg_id), -1) + visible = [(i, m) for i, m in enumerate(messages) if not getattr(m, "hidden", False)] + fates = age_tool_results([m for _, m in visible], cutoff_idx=next( + (v for v, (i, _) in enumerate(visible) if i == cutoff_idx), -1)) lines = [] - for m in messages: - if getattr(m, "hidden", False): - continue + for v, (i, m) in enumerate(visible): if m.role == "user": text = m.content if isinstance(m.content, str) else str(m.content) lines.append(f"User: {strip_forged_sentinels(clamp_recap_text(text))}") @@ -157,7 +161,13 @@ def build_history_prefix(messages, cutoff_msg_id: Optional[str] = None) -> str: elif m.role == "tool_call": lines.append(recap_tool_call_line(m.content)) elif m.role == "tool_result": - lines.append(recap_tool_result_line(m.content)) + body = fates.get(v) + if body is None: + lines.append(recap_tool_result_line(m.content)) + else: + tool_name = m.content.get("tool_name") if isinstance(m.content, dict) else None + label = f"Tool result ({tool_name})" if tool_name else "Tool result" + lines.append(f"{label}: {strip_forged_sentinels(body)}") if not lines: return "" return f"{SESSION_RECAP_OPEN}\n{PLATFORM_NOTE_PREAMBLE}\n" + "\n".join(lines) + f"\n{SESSION_RECAP_CLOSE}" diff --git a/backend/tests/test_aged_recap_lines.py b/backend/tests/test_aged_recap_lines.py new file mode 100644 index 00000000..f89c3ca2 --- /dev/null +++ b/backend/tests/test_aged_recap_lines.py @@ -0,0 +1,90 @@ +"""Pins the hermes-aging recap contract (ENG-354 endgame): duplicates collapse, the newest +results survive verbatim, old bulk becomes a re-runnable stub, pre-cutoff history ages +instead of vanishing, and the whole recap is bounded no matter how long the session ran.""" + +from backend.apps.agents.core.models import AgentSession, Message +from backend.apps.agents.manager.session.aged_recap_lines import ( + DUPLICATE_LINE, + PRUNE_MIN_CHARS, + TAIL_BUDGET_CHARS, + age_tool_results, + stub_line, +) +from backend.apps.agents.manager.session.history_compaction import build_history_prefix + + +def p_msgs(*pairs): + out = [] + for tool, args, result in pairs: + out.append(Message(role="tool_call", content={"tool": tool, "input": args})) + out.append(Message(role="tool_result", content={"text": result, "tool_name": tool})) + return out + + +def test_old_bulk_ages_into_a_rerunnable_stub(): + msgs = p_msgs(*[("Read", {"file_path": f"/tmp/f{i}.txt"}, f"body{i} " + "x" * 5000) for i in range(12)]) + fates = age_tool_results(msgs) + aged = [f for f in fates.values() if f.startswith("[Read]")] + assert aged, "no stubs produced" + assert any("/tmp/f0.txt" in f and "chars result)" in f for f in aged), "stub lost the re-runnable args" + + +def test_newest_results_survive_verbatim_within_budget(): + msgs = p_msgs(*[("Read", {"file_path": f"/f{i}"}, f"UNIQUE-{i} " + "y" * 3000) for i in range(10)]) + fates = age_tool_results(msgs) + newest = fates[len(msgs) - 1] + assert "UNIQUE-9" in newest and not newest.startswith("[Read]") + total_verbatim = sum(len(f) for f in fates.values() if not f.startswith("[")) + # The count floor is hermes semantics: at least TAIL_COUNT_FLOOR survive regardless of budget, each capped at TAIL_ITEM_CAP. + from backend.apps.agents.manager.session.aged_recap_lines import TAIL_COUNT_FLOOR, TAIL_ITEM_CAP + assert total_verbatim <= max(TAIL_BUDGET_CHARS + TAIL_ITEM_CAP, TAIL_COUNT_FLOOR * TAIL_ITEM_CAP) + assert len([f for f in fates.values() if not f.startswith("[")]) >= TAIL_COUNT_FLOOR + + +def test_duplicates_collapse_to_backreference(): + same = "identical result " + "z" * 400 + msgs = p_msgs(("Read", {"file_path": "/a"}, same), ("Read", {"file_path": "/a"}, same), ("Read", {"file_path": "/a"}, same)) + fates = age_tool_results(msgs) + dupes = [f for f in fates.values() if f == DUPLICATE_LINE] + assert len(dupes) == 2, "older duplicates must collapse; newest keeps the copy" + + +def test_small_results_always_survive_whole(): + msgs = p_msgs(*[("Bash", {"command": f"echo {i}"}, f"tiny-{i}") for i in range(30)]) + fates = age_tool_results(msgs) + assert all(f.startswith("tiny-") for f in fates.values()) + + +def test_precutoff_ages_instead_of_vanishing(): + s = AgentSession(name="t", model="sonnet") + for i in range(10): + s.messages.append(Message(role="tool_call", content={"tool": "Bash", "input": {"command": f"grep -rn pat{i} src/"}})) + s.messages.append(Message(role="tool_result", content={"text": f"match{i} " + "m" * 3000, "tool_name": "Bash"})) + cutoff = s.messages[9].id + recap = build_history_prefix(s.messages, cutoff_msg_id=cutoff) + assert "grep -rn pat1" in recap, "pre-cutoff command vanished; aging must keep the trail" + assert "match9" in recap, "newest result must survive verbatim" + pre = recap.split("match9")[0] + assert "chars result)" in pre, "pre-cutoff results must be stubs, not verbatim" + + +def test_recap_is_bounded_for_marathon_sessions(): + s = AgentSession(name="t", model="sonnet") + for i in range(120): + s.messages.append(Message(role="tool_call", content={"tool": "Read", "input": {"file_path": f"/tmp/m{i}.txt"}})) + s.messages.append(Message(role="tool_result", content={"text": f"MARA-{i} " + "w" * 8000, "tool_name": "Read"})) + recap = build_history_prefix(s.messages) + assert len(recap) < 80_000, f"recap unbounded: {len(recap)} chars for a 120-read session" + assert "MARA-119" in recap + assert "/tmp/m0.txt" in recap + + +def test_stub_format_matches_hermes_shape(): + line = stub_line("Bash", '{"command": "pytest -q"}', 22_770) + assert line == '[Bash] {"command": "pytest -q"} (22,770 chars result)' + + +def test_below_floor_never_stubbed(): + msgs = p_msgs(("Bash", {"command": "date"}, "x" * (PRUNE_MIN_CHARS - 1))) + fates = age_tool_results(msgs) + assert not list(fates.values())[0].startswith("[Bash]")