From fe2ae409832bba1fbab8153d6e1705ce4cd5d869 Mon Sep 17 00:00:00 2001 From: giggling-ginger <110955495+giggling-ginger@users.noreply.github.com> Date: Fri, 10 Jul 2026 08:11:21 +0000 Subject: [PATCH] fix(agent): demote oversized tool results in protected compression tail After multiple in-place compactions, short tool-heavy sessions can leave nearly every remaining message inside protect_last_n while those messages are huge completed file/tool outputs. The middle compress window then makes no material token progress and the turn dies with "Cannot compress further" (#61932). Cap the prune message floor at the same bound as tail-cut, and under pressure demote bulky protected-tail tool bodies (keeping a short recent floor) so preflight can reclaim headroom without wiping the active ask. --- agent/context_compressor.py | 180 +++++++++++---- .../test_protected_tail_pressure_61932.py | 207 ++++++++++++++++++ 2 files changed, 349 insertions(+), 38 deletions(-) create mode 100644 tests/agent/test_protected_tail_pressure_61932.py diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 9eaee872e353..aaa7c0ddb574 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -335,6 +335,11 @@ _ACTIVE_TASK_MAX_CHARS = 1400 # high for small/light tails, but using all 20 as a hard floor here would bring # back the old large-tool-output case where nothing can be compacted. _MAX_TAIL_MESSAGE_FLOOR = 8 +# Under context pressure (protected-tail tool bodies alone exceed the soft +# tail budget), demote large completed tool/file outputs even inside the +# protected region — but always keep this many trailing messages verbatim so +# the active user ask / latest tool pair remain readable. Issue #61932. +_PRESSURE_KEEP_RECENT_MESSAGES = 3 # Models with context windows below this get their compression threshold # floored at ``_SMALL_CTX_THRESHOLD_PERCENT`` (raise-only — an explicitly @@ -1918,7 +1923,14 @@ class ContextCompressor(ContextEngine): fall within ``protect_tail_tokens`` (when provided) OR the last ``protect_tail_count`` messages (backward-compatible default). When both are given, the token budget takes priority and the message - count acts as a hard minimum floor. + count acts as a hard minimum floor — capped at + ``_MAX_TAIL_MESSAGE_FLOOR`` so a default ``protect_last_n=20`` cannot + freeze a whole run of bulky tool outputs against pruning. + + When the protected region itself still exceeds the soft tail budget + (``protect_tail_tokens * 1.5``), a pressure pass demotes large + completed tool/file outputs *inside* that region while keeping a + short recent floor verbatim (issue #61932). Returns (pruned_messages, pruned_count). """ @@ -1946,10 +1958,17 @@ class ContextCompressor(ContextEngine): # Determine the prune boundary if protect_tail_tokens is not None and protect_tail_tokens > 0: - # Token-budget approach: walk backward accumulating tokens + # Token-budget approach: walk backward accumulating tokens. + # Cap the message-count floor the same way tail-cut does so a + # default protect_last_n=20 cannot lock a bulky recent tool run + # outside the compressible / prunable window (#61932). accumulated = 0 boundary = len(result) - min_protect = min(protect_tail_count, len(result)) + min_protect = min( + protect_tail_count, + len(result), + _MAX_TAIL_MESSAGE_FLOOR, + ) for i in range(len(result) - 1, -1, -1): msg = result[i] msg_tokens = _estimate_msg_budget_tokens(msg) @@ -1997,54 +2016,53 @@ class ContextCompressor(ContextEngine): else: content_hashes[h] = (i, msg.get("tool_call_id", "?")) - # Pass 2: Replace old tool results with informative summaries - for i in range(prune_boundary): - msg = result[i] + def _demote_tool_result_at(idx: int) -> bool: + """Replace a bulky tool result at ``idx`` with a 1-line summary. + + Returns True when the message was modified. + """ + nonlocal pruned + msg = result[idx] if msg.get("role") != "tool": - continue + return False content = msg.get("content", "") - # Multimodal content (base64 screenshots etc.): strip the image - # payload — keep a lightweight text placeholder in its place. - # Without this, an old computer_use screenshot (~1MB base64 + - # ~1500 real tokens) survives every compression pass forever. if isinstance(content, list): stripped = _strip_image_parts_from_parts(content) if stripped is not None: - result[i] = {**msg, "content": stripped} + result[idx] = {**msg, "content": stripped} pruned += 1 - continue + return True + return False if isinstance(content, dict) and content.get("_multimodal"): summary = content.get("text_summary") or "[screenshot removed to save context]" - result[i] = {**msg, "content": f"[screenshot removed] {summary[:200]}"} + result[idx] = {**msg, "content": f"[screenshot removed] {summary[:200]}"} pruned += 1 - continue + return True if not isinstance(content, str): - continue + return False if not content or content == _PRUNED_TOOL_PLACEHOLDER: - continue - # Skip already-deduplicated or previously-summarized results + return False if content.startswith("[Duplicate tool output"): - continue - # Only prune if the content is substantial (>200 chars) - if len(content) > 200: - call_id = msg.get("tool_call_id", "") - tool_name, tool_args = call_id_to_tool.get(call_id, ("unknown", "")) - summary = _summarize_tool_result(tool_name, tool_args, content) - result[i] = {**msg, "content": summary} - pruned += 1 + return False + # Already replaced by a prior prune/pressure pass (1-line summary). + if content.startswith("[") and " chars)" in content and len(content) < 400: + return False + if content.startswith("[screenshot removed"): + return False + if len(content) <= 200: + return False + call_id = msg.get("tool_call_id", "") + tool_name, tool_args = call_id_to_tool.get(call_id, ("unknown", "")) + summary = _summarize_tool_result(tool_name, tool_args, content) + result[idx] = {**msg, "content": summary} + pruned += 1 + return True - # Pass 3: Truncate large tool_call arguments in assistant messages - # outside the protected tail. write_file with 50KB content, for - # example, survives pruning entirely without this. - # - # The shrinking is done inside the parsed JSON structure so the - # result remains valid JSON — otherwise downstream providers 400 - # on every subsequent turn until the broken call falls out of - # the window. See ``_truncate_tool_call_args_json`` docstring. - for i in range(prune_boundary): - msg = result[i] + def _truncate_tool_call_args_at(idx: int) -> bool: + """Shrink large tool_call argument payloads at ``idx``.""" + msg = result[idx] if msg.get("role") != "assistant" or not msg.get("tool_calls"): - continue + return False new_tcs = [] modified = False for tc in msg["tool_calls"]: @@ -2057,7 +2075,93 @@ class ContextCompressor(ContextEngine): modified = True new_tcs.append(tc) if modified: - result[i] = {**msg, "tool_calls": new_tcs} + result[idx] = {**msg, "tool_calls": new_tcs} + return modified + + # Pass 2: Replace old tool results with informative summaries + for i in range(max(0, prune_boundary)): + _demote_tool_result_at(i) + + # Pass 3: Truncate large tool_call arguments in assistant messages + # outside the protected tail. write_file with 50KB content, for + # example, survives pruning entirely without this. + # + # The shrinking is done inside the parsed JSON structure so the + # result remains valid JSON — otherwise downstream providers 400 + # on every subsequent turn until the broken call falls out of + # the window. See ``_truncate_tool_call_args_json`` docstring. + for i in range(max(0, prune_boundary)): + _truncate_tool_call_args_at(i) + + # Pass 4 (issue #61932): protected-tail pressure demotion. + # After multiple in-place compactions the transcript can be short + # enough that nearly every remaining message sits inside the + # protected floor, yet those messages are huge completed tool / + # file outputs. Summarizing the (empty) middle does nothing and + # preflight ends in "Cannot compress further". Demote bulky tool + # bodies *inside* the protected region until the protected tail + # fits the soft budget, always keeping a short recent floor + # verbatim so the active ask stays readable. + if protect_tail_tokens is not None and protect_tail_tokens > 0 and result: + soft_ceiling = int(protect_tail_tokens * 1.5) + keep_recent = min(_PRESSURE_KEEP_RECENT_MESSAGES, len(result)) + demote_end = len(result) - keep_recent + + def _protected_region_tokens() -> int: + start = max(0, prune_boundary) + return sum( + _estimate_msg_budget_tokens(result[i]) + for i in range(start, len(result)) + ) + + if demote_end > prune_boundary and _protected_region_tokens() > soft_ceiling: + pressure_hits = 0 + for i in range(max(0, prune_boundary), demote_end): + if _demote_tool_result_at(i): + pressure_hits += 1 + if _truncate_tool_call_args_at(i): + pressure_hits += 1 + if _protected_region_tokens() <= soft_ceiling: + break + # If the short recent floor itself is still dominated by a + # stack of huge tool bodies, demote every protected tool + # result except the single most recent one. The active + # user message (usually the last row) stays untouched. + if _protected_region_tokens() > soft_ceiling: + last_tool_idx = None + for i in range(len(result) - 1, -1, -1): + if result[i].get("role") == "tool": + last_tool_idx = i + break + for i in range(max(0, prune_boundary), len(result)): + if last_tool_idx is not None and i == last_tool_idx: + continue + if result[i].get("role") == "tool": + if _demote_tool_result_at(i): + pressure_hits += 1 + elif result[i].get("role") == "assistant": + if _truncate_tool_call_args_at(i): + pressure_hits += 1 + # Absolute last resort: even the newest tool body can + # be larger than the soft budget alone (one 200KB file + # read). Summarize it so compression can still reclaim + # enough headroom to continue the session. + if ( + last_tool_idx is not None + and last_tool_idx >= prune_boundary + and _protected_region_tokens() > soft_ceiling + ): + if _demote_tool_result_at(last_tool_idx): + pressure_hits += 1 + if pressure_hits and not self.quiet_mode: + logger.info( + "Pre-compression pressure demotion: reclaimed protected-tail " + "tool output (%d change(s); protected region now ~%s tokens, " + "soft ceiling %s)", + pressure_hits, + f"{_protected_region_tokens():,}", + f"{soft_ceiling:,}", + ) return result, pruned diff --git a/tests/agent/test_protected_tail_pressure_61932.py b/tests/agent/test_protected_tail_pressure_61932.py new file mode 100644 index 000000000000..68f2068bbd28 --- /dev/null +++ b/tests/agent/test_protected_tail_pressure_61932.py @@ -0,0 +1,207 @@ +"""Algorithmic reproduction and regression for issue #61932. + +After several in-place compactions a tool-heavy session can be short enough +that nearly every remaining message sits inside the protected recent tail, +yet those messages are huge completed ``read_file`` / tool outputs. The +middle compress window is then empty or tiny, preflight makes no material +token progress, and the turn dies with:: + + Context length exceeded (174,833 tokens). Cannot compress further. + +This is the core compressor contract — not Desktop/Windows-specific. +""" + +from __future__ import annotations + +from unittest.mock import patch + +import pytest + +from agent.context_compressor import ( + ContextCompressor, + _MAX_TAIL_MESSAGE_FLOOR, + _PRESSURE_KEEP_RECENT_MESSAGES, +) +from agent.model_metadata import estimate_messages_tokens_rough +from agent.turn_context import _compression_made_progress + + +def _unique_tool_pair(i: int, chars: int) -> list[dict]: + """Assistant tool_call + unique tool result (no dedupe shortcut).""" + body = f"FILE_{i}_START\n" + (f"line {i} unique payload " * (chars // 22))[:chars] + return [ + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": f"call_{i}", + "type": "function", + "function": { + "name": "read_file", + "arguments": f'{{"path":"f{i}.py"}}', + }, + } + ], + }, + { + "role": "tool", + "tool_call_id": f"call_{i}", + "content": body, + }, + ] + + +def _already_compacted_session( + *, + n_pairs: int, + tool_chars: int, + user_chars: int, +) -> list[dict]: + """Shape after multiple in-place compactions: head + handoff + heavy tail.""" + msgs: list[dict] = [ + {"role": "system", "content": "You are Hermes."}, + {"role": "user", "content": "Investigate thoroughly"}, + {"role": "assistant", "content": "OK"}, + { + "role": "user", + "content": ( + "[CONTEXT COMPACTION — REFERENCE ONLY]\n" + + ("Prior findings. " * 200) + ), + }, + {"role": "assistant", "content": "Continuing from compacted context."}, + ] + for i in range(n_pairs): + msgs.extend(_unique_tool_pair(i, tool_chars)) + msgs.append( + { + "role": "user", + "content": "Full structured report:\n" + ("U" * user_chars), + } + ) + return msgs + + +@pytest.fixture() +def compressor_128k(): + with patch( + "agent.context_compressor.get_model_context_length", + return_value=128_000, + ): + c = ContextCompressor( + model="openai-codex/gpt-test", + threshold_percent=0.50, + summary_target_ratio=0.20, + protect_first_n=3, + protect_last_n=20, + quiet_mode=True, + config_context_length=128_000, + ) + c._generate_summary = lambda *a, **k: "compact summary of earlier investigation" + return c + + +class TestProtectedTailPressure61932: + def test_pressure_constants_aligned_with_tail_floor(self): + assert _MAX_TAIL_MESSAGE_FLOOR == 8 + assert _PRESSURE_KEEP_RECENT_MESSAGES == 3 + assert _PRESSURE_KEEP_RECENT_MESSAGES <= _MAX_TAIL_MESSAGE_FLOOR + + def test_prune_demotes_protected_tail_when_tool_bodies_dominate( + self, compressor_128k + ): + """Protected-region tool bodies that blow the soft budget must demote. + + With protect_last_n=20 and only ~14 messages, the old floor treated + every remaining tool result as sacred and prune was a pure no-op. + """ + c = compressor_128k + msgs = _already_compacted_session( + n_pairs=4, tool_chars=200_000, user_chars=80_000 + ) + before = estimate_messages_tokens_rough(msgs) + assert before > c.context_length + + pruned, n = c._prune_old_tool_results( + msgs, + protect_tail_count=c.protect_last_n, + protect_tail_tokens=c.tail_token_budget, + ) + after = estimate_messages_tokens_rough(pruned) + + assert n >= 1, "pressure demotion should touch at least one tool body" + assert after < before * 0.5, ( + f"expected large reclaim from protected-tail demotion: " + f"{before:,} → {after:,}" + ) + # Active user ask must remain verbatim. + assert pruned[-1]["role"] == "user" + assert pruned[-1]["content"].startswith("Full structured report:") + assert "U" * 100 in pruned[-1]["content"] + + def test_compress_escapes_cannot_compress_further_dead_end( + self, compressor_128k + ): + """Full compress path must materially reduce an over-context tail. + + Reproduces the #61932 failure class: multipass compression previously + dropped a couple of message rows while leaving ~170k tokens intact, + then reported no further progress. + """ + c = compressor_128k + msgs = _already_compacted_session( + n_pairs=4, tool_chars=200_000, user_chars=80_000 + ) + rough0 = estimate_messages_tokens_rough(msgs) + assert rough0 > c.context_length + + cur = msgs + tok = rough0 + last_progress = False + for _pass in range(3): + o_len, o_tok = len(cur), tok + out = c.compress(list(cur), current_tokens=tok) + n_tok = estimate_messages_tokens_rough(out) + last_progress = _compression_made_progress( + o_len, len(out), o_tok, n_tok + ) + cur, tok = out, n_tok + if n_tok < c.threshold_tokens and n_tok < c.context_length: + break + + assert tok < c.context_length, ( + f"still over context after compression: {tok:,} >= {c.context_length:,}" + ) + assert tok < rough0 * 0.5, ( + f"compression did not reclaim enough headroom: {rough0:,} → {tok:,}" + ) + # Either we recovered under threshold, or the last pass still made + # progress (never a pure no-op dead-end above the window). + assert tok < c.threshold_tokens or last_progress + + def test_light_tail_still_keeps_recent_tool_bodies(self, compressor_128k): + """Pressure demotion must not fire on a normal-sized protected tail.""" + c = compressor_128k + msgs = _already_compacted_session( + n_pairs=2, tool_chars=800, user_chars=200 + ) + before_tools = [ + m["content"] + for m in msgs + if m.get("role") == "tool" and isinstance(m.get("content"), str) + ] + pruned, n = c._prune_old_tool_results( + msgs, + protect_tail_count=c.protect_last_n, + protect_tail_tokens=c.tail_token_budget, + ) + after_tools = [ + m["content"] + for m in pruned + if m.get("role") == "tool" and isinstance(m.get("content"), str) + ] + assert after_tools, "expected tool messages to remain" + # Light unique bodies fit inside the soft budget — none should demote. + assert n == 0 + assert after_tools == before_tools