diff --git a/backend/apps/agents/manager/run/RunOptions.py b/backend/apps/agents/manager/run/RunOptions.py index 38d281bb..1b84664a 100644 --- a/backend/apps/agents/manager/run/RunOptions.py +++ b/backend/apps/agents/manager/run/RunOptions.py @@ -320,6 +320,31 @@ class RunOptions(AgentManagerProtocol): if distilled: fenced = wrap_platform_note(f"Summary of earlier conversation (older turns compacted):\n{distilled}") history = f"{fenced}\n\n{history}" if history else fenced + elif session.compacted_through_msg_id and p_mode != "none": + # A rebuild that drops old turns AND carries no gist of them is the agent's memory + # loss, and until now it happened in total silence: the distiller returns "" for five + # different reasons (feature off, cutoff gone from the branch, empty body, the aux + # call raising, the aux call returning nothing) and only one of them logs, at DEBUG. + # The recap never carries the model's own replies by design, so this summary IS the + # memory; without it the agent knows what was ASKED and which tools ran, but not what + # it concluded, which is exactly the "confidently describes work that isn't there" + # report. Say it out loud and stamp it, so the next report is diagnosable instead of + # unexplainable (the ENG-397 move). + logger.warning( + f"Agent {session_id}: rebuilding past a compaction cutoff with NO distilled " + f"summary; this turn loses the gist of its own earlier work (prefix={p_mode})" + ) + try: + from backend.apps.service.client import submit_diagnostic + submit_diagnostic({ + "kind": "recap_summary_missing", + "session_id": session_id, + "model": session.model, + "prefix_mode": p_mode, + "recap_chars": len(history or ""), + }) + except Exception: + logger.debug("submit_diagnostic recap_summary_missing failed", exc_info=True) if history: # SYSTEM channel, not the user message (ENG-358 structural fix): a transcript recap # inside user content is byte-for-byte what anti-distillation filters hunt, and no diff --git a/backend/tests/test_recap_summary_missing_is_loud.py b/backend/tests/test_recap_summary_missing_is_loud.py new file mode 100644 index 00000000..fbbe2785 --- /dev/null +++ b/backend/tests/test_recap_summary_missing_is_loud.py @@ -0,0 +1,60 @@ +"""The agent's memory loss had no signal at all, which is why it kept being unexplainable. + +A fresh-session rebuild sends the recap (the user's asks + the tool trail, never the model's own +replies, by design) plus a model-written distilled summary of the dropped span. That summary IS the +agent's memory of its own earlier conclusions. `distilled_history_summary` returns "" for FIVE +different reasons, and only one of them logged, at DEBUG: + + feature off | cutoff no longer on the branch | empty body | aux call raised | aux returned "" + +The aux call is a real LLM call, so it fails whenever aux is unavailable (observed live: +"No AI provider connected for auxiliary LLM call"). When it does, the rebuilt turn knows what was +ASKED and which tools ran, but not what it CONCLUDED, which is exactly the field report: the agent +confidently described work that did not exist, then retracted it after re-reading the workspace. + +Rule: a guard may never disable itself in silence. +""" +import inspect + +from backend.apps.agents.manager.run import RunOptions as RO +from backend.apps.agents.manager.session import distill_history + + +def p_src(): + return inspect.getsource(RO) + + +def test_a_rebuild_with_no_summary_warns_and_reports(): + src = p_src() + i = src.index('elif session.compacted_through_msg_id and p_mode != "none":') + block = src[i:i + 1600] + assert "logger.warning" in block, "silence is the bug; it must say so" + assert '"kind": "recap_summary_missing"' in block, "and it must be queryable on the fleet" + assert "session_id" in block, "ENG-397's lesson: an envelope with no session is untestable" + + +def test_it_only_fires_when_memory_was_actually_expected(): + """No cutoff means nothing was dropped, so nothing was lost; `none` means a policy block already + stripped the recap deliberately. Neither is the memory-loss case.""" + src = p_src() + i = src.index('elif session.compacted_through_msg_id and p_mode != "none":') + cond = src[i:i + 90] + assert "compacted_through_msg_id" in cond + assert 'p_mode != "none"' in cond + + +def test_it_sits_on_the_else_of_the_summary_being_present(): + """It must be the ELSE of `if distilled:`, or it would fire on healthy rebuilds too.""" + src = p_src() + assert src.index("if distilled:") < src.index('elif session.compacted_through_msg_id') + + +def test_the_distiller_really_has_five_silent_exits(): + """If this count ever changes, the comment above the warning is stale and should be re-read.""" + src = inspect.getsource(distill_history.distilled_history_summary) + assert src.count('return ""') == 5, f"expected 5 silent-empty exits, found {src.count('return ')}" + + +def test_the_feature_is_on_by_default(): + """A default-off distiller would make every long chat forget, which is worth knowing loudly.""" + assert distill_history.DISTILL_ENABLED is True