From 71a826c41d372259334e8d207325200afd98ef39 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Thu, 27 Aug 2026 12:06:07 -0700 Subject: [PATCH] [eric] agents: the mid-turn breaker says whose session it cannot protect instead of going quiet (ENG-391) --- backend/apps/agents/manager/context_budget.py | 16 ++++ .../apps/agents/manager/streaming/state.py | 3 + .../tests/test_midturn_breaker_liveness.py | 95 +++++++++++++++++++ 3 files changed, 114 insertions(+) create mode 100644 backend/tests/test_midturn_breaker_liveness.py diff --git a/backend/apps/agents/manager/context_budget.py b/backend/apps/agents/manager/context_budget.py index 824be19a..c75d8541 100644 --- a/backend/apps/agents/manager/context_budget.py +++ b/backend/apps/agents/manager/context_budget.py @@ -8,6 +8,7 @@ session.messages, the originals stay for the UI drawer and only the history sent is trimmed downstream (see backend/CLAUDE.md: "compaction must actually trim, not just mark").""" import os +import logging from typing import Dict, Optional from typeguard import typechecked @@ -17,6 +18,8 @@ from backend.apps.agents.core.ws_manager import ws_manager from backend.apps.agents.manager.session.history_compaction import get_branch_messages from backend.apps.agents.manager.streaming.state import TurnState +logger = logging.getLogger(__name__) + @typechecked def compact_ceiling_tokens(session: AgentSession) -> int: @@ -71,6 +74,19 @@ def maybe_break_midturn(session: AgentSession, turn: TurnState, msg_usage: Dict) except Exception: return False if total <= 0: + # No usage on this message. On the codex/GPT lane assistant messages NEVER carry usage + # (it arrives only on the ResultMessage, at turn end), so this is not a hiccup: the breaker + # is inert for that entire session and one giant turn can run to the context ceiling with + # nothing watching. A guard may never disable itself in silence, so it names what it just + # stopped protecting. Once per turn: this path runs on every assistant message. + if not turn.usage_absence_reported: + turn.usage_absence_reported = True + logger.warning( + "[context-break] session %s on model %s sends no per-message usage, so the " + "mid-turn context breaker cannot run for it; this turn is unprotected against a " + "single-turn context blowout (ENG-391)", + getattr(session, "id", "?"), getattr(session, "model", "?"), + ) return False # Keep the session's counter honest mid-turn: a broken turn never gets its ResultMessage accounting, and the next pre-send guard reads this. session.tokens["input"] = total diff --git a/backend/apps/agents/manager/streaming/state.py b/backend/apps/agents/manager/streaming/state.py index 77afc4fd..e716f3d4 100644 --- a/backend/apps/agents/manager/streaming/state.py +++ b/backend/apps/agents/manager/streaming/state.py @@ -66,5 +66,8 @@ class TurnState(BaseModel): # Mid-turn context breaker: fires once per turn, and only after a below-trigger reading (a turn that STARTS over the trigger must run, or a failed shrink would break-loop forever). context_break_fired: bool = False saw_input_below_trigger: bool = False + # Said once per turn when the provider sends no usage at all, so the breaker being + # structurally inert on that lane is visible instead of silent (ENG-391). + usage_absence_reported: bool = False # The LAST inference step's request size (input + cache read + cache creation): the true live context. The ResultMessage's usage sums these across every step of the turn, which is billing, not context. last_step_input: int = 0 diff --git a/backend/tests/test_midturn_breaker_liveness.py b/backend/tests/test_midturn_breaker_liveness.py new file mode 100644 index 00000000..e5980af1 --- /dev/null +++ b/backend/tests/test_midturn_breaker_liveness.py @@ -0,0 +1,95 @@ +"""The mid-turn context breaker must FIRE, and must say so when it structurally cannot. + +ENG-391: on the codex/GPT lane assistant messages carry no usage at all (it arrives only on the +ResultMessage, at turn end), so `maybe_break_midturn` bailed on its first line for every message of +every GPT session. Its tests passed the whole time, because they only proved it does not crash. + +That is the row-6 shape CLAUDE.md names: present, reachable, doing nothing, with nothing saying so. +A guard that never fires is indistinguishable from one that was never needed. +""" + +import logging + +import pytest + +from backend.apps.agents.core.models import AgentSession +from backend.apps.agents.manager import context_budget as p_cb +from backend.apps.agents.manager.streaming.state import TurnState + + +@pytest.fixture +def p_logs(): + """Read context_budget's own records: backend/main.py sets propagate=False on `backend`, so + caplog goes silent here the moment any test has imported the app.""" + rec: list = [] + + class P_Sink(logging.Handler): + def emit(self, r) -> None: + rec.append(r) + + lg = logging.getLogger("backend.apps.agents.manager.context_budget") + h = P_Sink() + lg.addHandler(h) + prev = lg.level + lg.setLevel(logging.DEBUG) + try: + yield rec + finally: + lg.removeHandler(h) + lg.setLevel(prev) + + +def p_session() -> AgentSession: + s = AgentSession(id="s-gpt", name="c", title="c", model="cx/gpt-5.6") + s.context_window = 200_000 + return s + + +def test_it_actually_FIRES_when_usage_crosses_the_trigger(): + """The liveness assertion the original tests never made.""" + s, t = p_session(), TurnState() + trigger = p_cb.compact_trigger_tokens(s) + assert p_cb.maybe_break_midturn(s, t, {"input_tokens": 10}) is False, "a low reading arms it" + assert t.saw_input_below_trigger is True + assert p_cb.maybe_break_midturn(s, t, {"input_tokens": trigger + 1}) is True, \ + "crossing the trigger mid-turn MUST break the turn" + assert s.pending_continuation is True and s.needs_fresh_session is True + + +def test_it_fires_only_once_per_turn(): + s, t = p_session(), TurnState() + trigger = p_cb.compact_trigger_tokens(s) + p_cb.maybe_break_midturn(s, t, {"input_tokens": 10}) + assert p_cb.maybe_break_midturn(s, t, {"input_tokens": trigger + 1}) is True + assert p_cb.maybe_break_midturn(s, t, {"input_tokens": trigger + 9999}) is False + + +def test_a_turn_that_STARTS_over_the_trigger_is_left_alone(): + """A failed shrink would break-loop forever otherwise.""" + s, t = p_session(), TurnState() + assert p_cb.maybe_break_midturn(s, t, {"input_tokens": p_cb.compact_trigger_tokens(s) + 1}) is False + + +def test_no_usage_at_all_is_ANNOUNCED_not_swallowed(p_logs): + """The GPT lane. It cannot run; it must say whose session it just stopped protecting.""" + s, t = p_session(), TurnState() + assert p_cb.maybe_break_midturn(s, t, {}) is False + said = " ".join(r.getMessage() for r in p_logs) + assert "cannot run" in said + assert "s-gpt" in said and "cx/gpt-5.6" in said, "name the session and the lane, not just the class" + assert "ENG-391" in said + + +def test_the_announcement_is_once_per_turn_not_per_message(p_logs): + """It runs on EVERY assistant message; a per-message warning would be its own bug.""" + s, t = p_session(), TurnState() + for _ in range(12): + p_cb.maybe_break_midturn(s, t, {}) + assert len([r for r in p_logs if "cannot run" in r.getMessage()]) == 1 + + +def test_a_lane_WITH_usage_never_triggers_the_warning(p_logs): + """The innocent case: Anthropic sends usage, so nothing is inert and nothing should be said.""" + s, t = p_session(), TurnState() + p_cb.maybe_break_midturn(s, t, {"input_tokens": 500}) + assert not [r for r in p_logs if "cannot run" in r.getMessage()]