mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
feat: require runway before prune-only compaction
Make prune-first compression cache-aware by only accepting prune-only compaction when it gets comfortably below threshold. If pruning merely dips under threshold, fall through to the existing summary compaction so we avoid frequent near-threshold recompressions. Tests cover both the conservative fallback and the prune-only fast path.
This commit is contained in:
parent
55729670be
commit
a1a90f3f10
2 changed files with 62 additions and 1 deletions
|
|
@ -74,6 +74,8 @@ class ContextCompressor:
|
|||
self.summary_model = summary_model_override or ""
|
||||
self._prune_protect_tokens = _adaptive_prune_protect(self.context_length)
|
||||
self._prune_minimum_tokens = _adaptive_prune_minimum(self.context_length)
|
||||
self._prune_runway_tokens = max(self._prune_minimum_tokens, int(self.threshold_tokens * 0.15))
|
||||
self._prune_target_tokens = max(0, self.threshold_tokens - self._prune_runway_tokens)
|
||||
|
||||
def update_from_response(self, usage: Dict[str, Any]):
|
||||
"""Update tracked token usage from API response."""
|
||||
|
|
@ -354,7 +356,7 @@ Write only the summary, starting with "[CONTEXT SUMMARY]:" prefix."""
|
|||
f" ✂️ Phase 1 (prune): removed {chars_saved:,} chars of old tool outputs "
|
||||
f"(~{tokens_saved_phase1:,} tokens saved)"
|
||||
)
|
||||
if pruned_tokens < self.threshold_tokens:
|
||||
if pruned_tokens <= self._prune_target_tokens:
|
||||
self.compression_count += 1
|
||||
pruned_messages = self._sanitize_tool_pairs(pruned_messages)
|
||||
if not self.quiet_mode:
|
||||
|
|
@ -364,6 +366,11 @@ Write only the summary, starting with "[CONTEXT SUMMARY]:" prefix."""
|
|||
)
|
||||
print(f" 💡 Compression #{self.compression_count} complete (prune only — no LLM call needed)")
|
||||
return pruned_messages
|
||||
if not self.quiet_mode and pruned_tokens < self.threshold_tokens:
|
||||
print(
|
||||
f" ↪️ Phase 1 recovered tokens but not enough runway "
|
||||
f"({pruned_tokens:,} > target {self._prune_target_tokens:,}); continuing to compaction"
|
||||
)
|
||||
messages = pruned_messages
|
||||
n_messages = len(messages)
|
||||
compress_start = self.protect_first_n
|
||||
|
|
|
|||
|
|
@ -400,3 +400,57 @@ class TestPruneToolOutputs:
|
|||
|
||||
assert pruned[-2]["content"] == huge_content
|
||||
assert pruned[-1]["content"] == "latest"
|
||||
|
||||
|
||||
class TestPruneAcceptancePolicy:
|
||||
def _make_compressor(self, *, context_length=128000):
|
||||
with patch("agent.context_compressor.get_model_context_length", return_value=context_length):
|
||||
return ContextCompressor(
|
||||
model="test/model",
|
||||
threshold_percent=0.50,
|
||||
protect_first_n=2,
|
||||
protect_last_n=1,
|
||||
quiet_mode=True,
|
||||
)
|
||||
|
||||
def test_prune_near_threshold_still_falls_back_to_summary(self):
|
||||
c = self._make_compressor()
|
||||
huge_content = "x" * 180000
|
||||
messages = [
|
||||
{"role": "system", "content": "sys"},
|
||||
{"role": "user", "content": "task"},
|
||||
{"role": "assistant", "content": "older"},
|
||||
{"role": "tool", "content": huge_content, "name": "terminal"},
|
||||
{"role": "assistant", "content": "newer"},
|
||||
{"role": "tool", "content": huge_content, "name": "terminal"},
|
||||
{"role": "assistant", "content": "tail"},
|
||||
]
|
||||
mock_response = MagicMock()
|
||||
mock_response.choices = [MagicMock()]
|
||||
mock_response.choices[0].message.content = "[CONTEXT SUMMARY]: compacted"
|
||||
|
||||
with patch("agent.context_compressor.estimate_messages_tokens_rough", return_value=62000), \
|
||||
patch("agent.context_compressor.call_llm", return_value=mock_response):
|
||||
result = c.compress(messages, current_tokens=68000)
|
||||
|
||||
assert any("CONTEXT SUMMARY" in (msg.get("content") or "") for msg in result)
|
||||
|
||||
def test_prune_only_is_allowed_when_it_buys_real_runway(self):
|
||||
c = self._make_compressor()
|
||||
huge_content = "x" * 180000
|
||||
messages = [
|
||||
{"role": "system", "content": "sys"},
|
||||
{"role": "user", "content": "task"},
|
||||
{"role": "assistant", "content": "older"},
|
||||
{"role": "tool", "content": huge_content, "name": "terminal"},
|
||||
{"role": "assistant", "content": "newer"},
|
||||
{"role": "tool", "content": huge_content, "name": "terminal"},
|
||||
{"role": "assistant", "content": "tail"},
|
||||
]
|
||||
|
||||
with patch("agent.context_compressor.estimate_messages_tokens_rough", return_value=48000), \
|
||||
patch.object(ContextCompressor, "_generate_summary", side_effect=AssertionError("summary should not be called")):
|
||||
result = c.compress(messages, current_tokens=68000)
|
||||
|
||||
assert result[3]["content"].startswith("[Tool output pruned")
|
||||
assert result[5]["content"] == huge_content
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue