mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-30 21:44:50 +02:00
[eric] agents: the mid-turn breaker says whose session it cannot protect instead of going quiet (ENG-391)
This commit is contained in:
@@ -8,6 +8,7 @@ session.messages, the originals stay for the UI drawer and only the history sent
|
||||
is trimmed downstream (see backend/CLAUDE.md: "compaction must actually trim, not just mark")."""
|
||||
|
||||
import os
|
||||
import logging
|
||||
from typing import Dict, Optional
|
||||
|
||||
from typeguard import typechecked
|
||||
@@ -17,6 +18,8 @@ from backend.apps.agents.core.ws_manager import ws_manager
|
||||
from backend.apps.agents.manager.session.history_compaction import get_branch_messages
|
||||
from backend.apps.agents.manager.streaming.state import TurnState
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@typechecked
|
||||
def compact_ceiling_tokens(session: AgentSession) -> int:
|
||||
@@ -71,6 +74,19 @@ def maybe_break_midturn(session: AgentSession, turn: TurnState, msg_usage: Dict)
|
||||
except Exception:
|
||||
return False
|
||||
if total <= 0:
|
||||
# No usage on this message. On the codex/GPT lane assistant messages NEVER carry usage
|
||||
# (it arrives only on the ResultMessage, at turn end), so this is not a hiccup: the breaker
|
||||
# is inert for that entire session and one giant turn can run to the context ceiling with
|
||||
# nothing watching. A guard may never disable itself in silence, so it names what it just
|
||||
# stopped protecting. Once per turn: this path runs on every assistant message.
|
||||
if not turn.usage_absence_reported:
|
||||
turn.usage_absence_reported = True
|
||||
logger.warning(
|
||||
"[context-break] session %s on model %s sends no per-message usage, so the "
|
||||
"mid-turn context breaker cannot run for it; this turn is unprotected against a "
|
||||
"single-turn context blowout (ENG-391)",
|
||||
getattr(session, "id", "?"), getattr(session, "model", "?"),
|
||||
)
|
||||
return False
|
||||
# Keep the session's counter honest mid-turn: a broken turn never gets its ResultMessage accounting, and the next pre-send guard reads this.
|
||||
session.tokens["input"] = total
|
||||
|
||||
@@ -66,5 +66,8 @@ class TurnState(BaseModel):
|
||||
# Mid-turn context breaker: fires once per turn, and only after a below-trigger reading (a turn that STARTS over the trigger must run, or a failed shrink would break-loop forever).
|
||||
context_break_fired: bool = False
|
||||
saw_input_below_trigger: bool = False
|
||||
# Said once per turn when the provider sends no usage at all, so the breaker being
|
||||
# structurally inert on that lane is visible instead of silent (ENG-391).
|
||||
usage_absence_reported: bool = False
|
||||
# The LAST inference step's request size (input + cache read + cache creation): the true live context. The ResultMessage's usage sums these across every step of the turn, which is billing, not context.
|
||||
last_step_input: int = 0
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
"""The mid-turn context breaker must FIRE, and must say so when it structurally cannot.
|
||||
|
||||
ENG-391: on the codex/GPT lane assistant messages carry no usage at all (it arrives only on the
|
||||
ResultMessage, at turn end), so `maybe_break_midturn` bailed on its first line for every message of
|
||||
every GPT session. Its tests passed the whole time, because they only proved it does not crash.
|
||||
|
||||
That is the row-6 shape CLAUDE.md names: present, reachable, doing nothing, with nothing saying so.
|
||||
A guard that never fires is indistinguishable from one that was never needed.
|
||||
"""
|
||||
|
||||
import logging
|
||||
|
||||
import pytest
|
||||
|
||||
from backend.apps.agents.core.models import AgentSession
|
||||
from backend.apps.agents.manager import context_budget as p_cb
|
||||
from backend.apps.agents.manager.streaming.state import TurnState
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def p_logs():
|
||||
"""Read context_budget's own records: backend/main.py sets propagate=False on `backend`, so
|
||||
caplog goes silent here the moment any test has imported the app."""
|
||||
rec: list = []
|
||||
|
||||
class P_Sink(logging.Handler):
|
||||
def emit(self, r) -> None:
|
||||
rec.append(r)
|
||||
|
||||
lg = logging.getLogger("backend.apps.agents.manager.context_budget")
|
||||
h = P_Sink()
|
||||
lg.addHandler(h)
|
||||
prev = lg.level
|
||||
lg.setLevel(logging.DEBUG)
|
||||
try:
|
||||
yield rec
|
||||
finally:
|
||||
lg.removeHandler(h)
|
||||
lg.setLevel(prev)
|
||||
|
||||
|
||||
def p_session() -> AgentSession:
|
||||
s = AgentSession(id="s-gpt", name="c", title="c", model="cx/gpt-5.6")
|
||||
s.context_window = 200_000
|
||||
return s
|
||||
|
||||
|
||||
def test_it_actually_FIRES_when_usage_crosses_the_trigger():
|
||||
"""The liveness assertion the original tests never made."""
|
||||
s, t = p_session(), TurnState()
|
||||
trigger = p_cb.compact_trigger_tokens(s)
|
||||
assert p_cb.maybe_break_midturn(s, t, {"input_tokens": 10}) is False, "a low reading arms it"
|
||||
assert t.saw_input_below_trigger is True
|
||||
assert p_cb.maybe_break_midturn(s, t, {"input_tokens": trigger + 1}) is True, \
|
||||
"crossing the trigger mid-turn MUST break the turn"
|
||||
assert s.pending_continuation is True and s.needs_fresh_session is True
|
||||
|
||||
|
||||
def test_it_fires_only_once_per_turn():
|
||||
s, t = p_session(), TurnState()
|
||||
trigger = p_cb.compact_trigger_tokens(s)
|
||||
p_cb.maybe_break_midturn(s, t, {"input_tokens": 10})
|
||||
assert p_cb.maybe_break_midturn(s, t, {"input_tokens": trigger + 1}) is True
|
||||
assert p_cb.maybe_break_midturn(s, t, {"input_tokens": trigger + 9999}) is False
|
||||
|
||||
|
||||
def test_a_turn_that_STARTS_over_the_trigger_is_left_alone():
|
||||
"""A failed shrink would break-loop forever otherwise."""
|
||||
s, t = p_session(), TurnState()
|
||||
assert p_cb.maybe_break_midturn(s, t, {"input_tokens": p_cb.compact_trigger_tokens(s) + 1}) is False
|
||||
|
||||
|
||||
def test_no_usage_at_all_is_ANNOUNCED_not_swallowed(p_logs):
|
||||
"""The GPT lane. It cannot run; it must say whose session it just stopped protecting."""
|
||||
s, t = p_session(), TurnState()
|
||||
assert p_cb.maybe_break_midturn(s, t, {}) is False
|
||||
said = " ".join(r.getMessage() for r in p_logs)
|
||||
assert "cannot run" in said
|
||||
assert "s-gpt" in said and "cx/gpt-5.6" in said, "name the session and the lane, not just the class"
|
||||
assert "ENG-391" in said
|
||||
|
||||
|
||||
def test_the_announcement_is_once_per_turn_not_per_message(p_logs):
|
||||
"""It runs on EVERY assistant message; a per-message warning would be its own bug."""
|
||||
s, t = p_session(), TurnState()
|
||||
for _ in range(12):
|
||||
p_cb.maybe_break_midturn(s, t, {})
|
||||
assert len([r for r in p_logs if "cannot run" in r.getMessage()]) == 1
|
||||
|
||||
|
||||
def test_a_lane_WITH_usage_never_triggers_the_warning(p_logs):
|
||||
"""The innocent case: Anthropic sends usage, so nothing is inert and nothing should be said."""
|
||||
s, t = p_session(), TurnState()
|
||||
p_cb.maybe_break_midturn(s, t, {"input_tokens": 500})
|
||||
assert not [r for r in p_logs if "cannot run" in r.getMessage()]
|
||||
Reference in New Issue
Block a user