From a1a90f3f10893c20ae94cbf89b7d534f378d520e Mon Sep 17 00:00:00 2001 From: teknium1 Date: Fri, 13 Mar 2026 21:46:09 -0700 Subject: [PATCH] feat: require runway before prune-only compaction Make prune-first compression cache-aware by only accepting prune-only compaction when it gets comfortably below threshold. If pruning merely dips under threshold, fall through to the existing summary compaction so we avoid frequent near-threshold recompressions. Tests cover both the conservative fallback and the prune-only fast path. --- agent/context_compressor.py | 9 ++++- tests/agent/test_context_compressor.py | 54 ++++++++++++++++++++++++++ 2 files changed, 62 insertions(+), 1 deletion(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 18990227513..1e0594dda16 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -74,6 +74,8 @@ class ContextCompressor: self.summary_model = summary_model_override or "" self._prune_protect_tokens = _adaptive_prune_protect(self.context_length) self._prune_minimum_tokens = _adaptive_prune_minimum(self.context_length) + self._prune_runway_tokens = max(self._prune_minimum_tokens, int(self.threshold_tokens * 0.15)) + self._prune_target_tokens = max(0, self.threshold_tokens - self._prune_runway_tokens) def update_from_response(self, usage: Dict[str, Any]): """Update tracked token usage from API response.""" @@ -354,7 +356,7 @@ Write only the summary, starting with "[CONTEXT SUMMARY]:" prefix.""" f" ✂️ Phase 1 (prune): removed {chars_saved:,} chars of old tool outputs " f"(~{tokens_saved_phase1:,} tokens saved)" ) - if pruned_tokens < self.threshold_tokens: + if pruned_tokens <= self._prune_target_tokens: self.compression_count += 1 pruned_messages = self._sanitize_tool_pairs(pruned_messages) if not self.quiet_mode: @@ -364,6 +366,11 @@ Write only the summary, starting with "[CONTEXT SUMMARY]:" prefix.""" ) print(f" 💡 Compression #{self.compression_count} complete (prune only — no LLM call needed)") return pruned_messages + if not self.quiet_mode and pruned_tokens < self.threshold_tokens: + print( + f" ↪️ Phase 1 recovered tokens but not enough runway " + f"({pruned_tokens:,} > target {self._prune_target_tokens:,}); continuing to compaction" + ) messages = pruned_messages n_messages = len(messages) compress_start = self.protect_first_n diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index 8f74c498f7a..eddf63a6cce 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -400,3 +400,57 @@ class TestPruneToolOutputs: assert pruned[-2]["content"] == huge_content assert pruned[-1]["content"] == "latest" + + +class TestPruneAcceptancePolicy: + def _make_compressor(self, *, context_length=128000): + with patch("agent.context_compressor.get_model_context_length", return_value=context_length): + return ContextCompressor( + model="test/model", + threshold_percent=0.50, + protect_first_n=2, + protect_last_n=1, + quiet_mode=True, + ) + + def test_prune_near_threshold_still_falls_back_to_summary(self): + c = self._make_compressor() + huge_content = "x" * 180000 + messages = [ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "task"}, + {"role": "assistant", "content": "older"}, + {"role": "tool", "content": huge_content, "name": "terminal"}, + {"role": "assistant", "content": "newer"}, + {"role": "tool", "content": huge_content, "name": "terminal"}, + {"role": "assistant", "content": "tail"}, + ] + mock_response = MagicMock() + mock_response.choices = [MagicMock()] + mock_response.choices[0].message.content = "[CONTEXT SUMMARY]: compacted" + + with patch("agent.context_compressor.estimate_messages_tokens_rough", return_value=62000), \ + patch("agent.context_compressor.call_llm", return_value=mock_response): + result = c.compress(messages, current_tokens=68000) + + assert any("CONTEXT SUMMARY" in (msg.get("content") or "") for msg in result) + + def test_prune_only_is_allowed_when_it_buys_real_runway(self): + c = self._make_compressor() + huge_content = "x" * 180000 + messages = [ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "task"}, + {"role": "assistant", "content": "older"}, + {"role": "tool", "content": huge_content, "name": "terminal"}, + {"role": "assistant", "content": "newer"}, + {"role": "tool", "content": huge_content, "name": "terminal"}, + {"role": "assistant", "content": "tail"}, + ] + + with patch("agent.context_compressor.estimate_messages_tokens_rough", return_value=48000), \ + patch.object(ContextCompressor, "_generate_summary", side_effect=AssertionError("summary should not be called")): + result = c.compress(messages, current_tokens=68000) + + assert result[3]["content"].startswith("[Tool output pruned") + assert result[5]["content"] == huge_content