mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
Surface warning when context exceeds compression threshold but compression is blocked
Previously, when a session crossed the compression threshold but compression was skipped (summary-LLM cooldown, #11529, or anti-thrashing, #40803), the model kept accumulating context until it hit the hard provider token limit and silently stopped answering — with no signal to the user about why. Changes: - context_compressor.should_compress_info() returns a (should_compress, reason) tuple. reason is 'cooldown:<seconds>' or 'ineffective' when compression is needed but blocked. should_compress() keeps its bool contract so existing callers (conversation_loop.py) and regression #29335 are unaffected. - turn_context.build_turn_context() emits a deduped _emit_warning when the context is over threshold but compression is blocked, advising /new or /compress. Dedup keys on the block *kind* (cooldown/ineffective), not the ticking countdown, so a cooldown doesn't re-fire the warning every turn. - Adds tests/agent/test_turn_context_overflow_warning.py covering the tuple shape, both block kinds, dedup, and re-fire-after-clear.
This commit is contained in:
parent
781968be5e
commit
b0a88899bf
4 changed files with 308 additions and 3 deletions
|
|
@ -1901,14 +1901,72 @@ class ContextCompressor(ContextEngine):
|
|||
def should_compress(self, prompt_tokens: int = None) -> bool:
|
||||
"""Check if context exceeds the compression threshold.
|
||||
|
||||
Returns ``True`` when compression should run now. For the caller-facing
|
||||
*reason* (e.g. why compression is skipped while still over threshold),
|
||||
see :meth:`should_compress_info`, which returns a ``(bool, reason)``
|
||||
tuple without changing the decision logic here.
|
||||
|
||||
Includes anti-thrashing protection: if the last two compressions
|
||||
each saved less than 10%, skip compression to avoid infinite loops
|
||||
where each pass removes only 1-2 messages.
|
||||
"""
|
||||
decision, _reason = self.should_compress_info(prompt_tokens)
|
||||
return decision
|
||||
|
||||
def should_compress_info(
|
||||
self, prompt_tokens: int = None
|
||||
) -> "tuple[bool, str | None]":
|
||||
"""Check if context exceeds the compression threshold.
|
||||
|
||||
Returns a ``(should_compress, reason)`` tuple instead of a bare bool so
|
||||
callers can tell *why* compression is skipped when it is skipped while
|
||||
the context is already over threshold. ``reason`` is ``None`` unless
|
||||
compression is needed but blocked:
|
||||
|
||||
* ``"cooldown:<seconds>"`` — the summary LLM is recovering from a
|
||||
recent 429/transient failure; compression is deferred to avoid the
|
||||
freeze loop described in #11529.
|
||||
* ``"ineffective"`` — anti-thrashing has backed off because the last
|
||||
two compressions each saved <10%.
|
||||
|
||||
When ``reason`` is non-``None`` the session is over its compression
|
||||
threshold yet cannot shrink — callers should surface a warning so the
|
||||
user knows the model may silently stop answering (the context keeps
|
||||
growing until it hits the hard provider limit). Without this signal an
|
||||
over-threshold session fails opaquely.
|
||||
|
||||
Includes anti-thrashing protection: if the last two compressions
|
||||
each saved less than 10%, skip compression to avoid infinite loops
|
||||
where each pass removes only 1-2 messages.
|
||||
"""
|
||||
tokens = prompt_tokens if prompt_tokens is not None else self.last_prompt_tokens
|
||||
if tokens < self.threshold_tokens:
|
||||
return False
|
||||
return not self._automatic_compression_blocked()
|
||||
return False, None
|
||||
if self._automatic_compression_blocked():
|
||||
return False, self._compression_block_reason() or "blocked"
|
||||
return True, None
|
||||
|
||||
def _compression_block_reason(self) -> "str | None":
|
||||
"""Return a human-readable reason for the current automatic-compaction
|
||||
block, derived from the same in-memory state that
|
||||
:meth:`_automatic_compression_blocked_locally` evaluates.
|
||||
|
||||
* ``"cooldown:<seconds>"`` — the summary LLM is recovering from a
|
||||
recent 429/transient failure; compression is deferred to avoid the
|
||||
freeze loop described in #11529.
|
||||
* ``"ineffective"`` — anti-thrashing has backed off (the last two
|
||||
compressions each saved <10%, or the fallback streak tripped).
|
||||
* ``None`` — no block active.
|
||||
"""
|
||||
_cooldown_remaining = self._summary_failure_cooldown_until - time.monotonic()
|
||||
if _cooldown_remaining > 0:
|
||||
return f"cooldown:{_cooldown_remaining:.0f}"
|
||||
if (
|
||||
self._ineffective_compression_count >= 2
|
||||
or self._fallback_compression_streak >= 2
|
||||
):
|
||||
return "ineffective"
|
||||
return None
|
||||
|
||||
def _refresh_durable_guards(self) -> None:
|
||||
"""Re-read durable cooldown + breaker state from the DB.
|
||||
|
|
|
|||
|
|
@ -750,6 +750,8 @@ def build_turn_context(
|
|||
lambda: None,
|
||||
)()
|
||||
|
||||
_should_compress_now = False
|
||||
_compress_block_reason = None
|
||||
if _preflight_deferred:
|
||||
logger.info(
|
||||
"Skipping preflight compression: rough estimate ~%s >= %s, "
|
||||
|
|
@ -765,13 +767,24 @@ def build_turn_context(
|
|||
int(_compression_cooldown.get("remaining_seconds", 0.0)),
|
||||
agent.session_id or "none",
|
||||
)
|
||||
# Context is over threshold but compression is blocked by the
|
||||
# summary-LLM cooldown — surface a warning (see block below).
|
||||
_cooldown_secs = _compression_cooldown.get("remaining_seconds", 0.0)
|
||||
_compress_block_reason = f"cooldown:{_cooldown_secs:.0f}"
|
||||
elif _codex_native_auto:
|
||||
logger.info(
|
||||
"Skipping Hermes preflight compression for codex app-server "
|
||||
"(mode=%s); Hermes will not start thread compaction here.",
|
||||
getattr(agent, "codex_app_server_auto_compaction", "native"),
|
||||
)
|
||||
elif _compressor.should_compress(_preflight_tokens):
|
||||
else:
|
||||
_should_compress_now = _compressor.should_compress(_preflight_tokens)
|
||||
if not _should_compress_now:
|
||||
# Context is over threshold but compression is blocked
|
||||
# (summary-LLM cooldown or anti-thrashing). Ask should_compress_info
|
||||
# for the human-readable reason so we can surface a warning below.
|
||||
_compress_block_reason = _compressor.should_compress_info(_preflight_tokens)[1]
|
||||
if _should_compress_now:
|
||||
_preflight_compressed = True
|
||||
logger.info(
|
||||
"Preflight compression: ~%s tokens >= %s threshold (model %s, ctx %s)",
|
||||
|
|
@ -844,7 +857,31 @@ def build_turn_context(
|
|||
f"{_preflight_tokens:,}",
|
||||
)
|
||||
break
|
||||
elif _compress_block_reason:
|
||||
# Context is already over the compression threshold, but compression
|
||||
# is blocked (summary LLM cooldown or anti-thrashing). Without a
|
||||
# signal the session keeps growing until the model silently stops
|
||||
# answering — the conversation hits the hard provider token limit
|
||||
# with no explanation. Surface a deduped warning so the user can
|
||||
# take action (/new or /compress) instead of hitting a silent hang.
|
||||
# Dedup on the *kind* of block (cooldown / ineffective), not the
|
||||
# exact countdown, so a cooldown ticking down 30→29→… doesn't re-fire
|
||||
# the warning every turn.
|
||||
_warn_kind = _compress_block_reason.split(":", 1)[0]
|
||||
_warn_key = ("ctx_overflow_blocked", _warn_kind)
|
||||
if getattr(agent, "_last_ctx_overflow_warn", None) != _warn_key:
|
||||
agent._last_ctx_overflow_warn = _warn_key
|
||||
agent._emit_warning(
|
||||
f"⚠ Context is over the compression threshold "
|
||||
f"(~{_preflight_tokens:,} tokens >= {_compressor.threshold_tokens:,}) "
|
||||
f"but compression is currently blocked ({_compress_block_reason}). "
|
||||
f"The model may stop responding. Run /new to start a fresh session "
|
||||
f"or /compress to retry immediately."
|
||||
)
|
||||
else:
|
||||
# Sub-threshold and unblocked — allow the overflow warning to fire
|
||||
# again next time the context is over threshold but blocked.
|
||||
agent._last_ctx_overflow_warn = None
|
||||
# ── Engine-driven sub-threshold preflight maintenance (#20316) ──
|
||||
# None of the threshold-path branches fired (not deferred, no
|
||||
# failure cooldown, not codex-native, and should_compress() said
|
||||
|
|
|
|||
|
|
@ -67,6 +67,7 @@ class _FakeAgent:
|
|||
self._todo_store = _FakeTodoStore()
|
||||
self._tool_guardrails = _FakeGuardrails()
|
||||
self._compression_warning = None
|
||||
self._emit_warning = MagicMock()
|
||||
self._interrupt_requested = False
|
||||
self._memory_write_origin = "assistant_tool"
|
||||
self._stream_context_scrubber = None
|
||||
|
|
|
|||
209
tests/agent/test_turn_context_overflow_warning.py
Normal file
209
tests/agent/test_turn_context_overflow_warning.py
Normal file
|
|
@ -0,0 +1,209 @@
|
|||
"""Tests for the silent-context-overflow warning (the fix for the bug where a
|
||||
session crosses the compression threshold but compression is blocked — by the
|
||||
summary-LLM cooldown (#11529) or anti-thrashing (#40803) — and the model then
|
||||
silently stops answering because nothing tells the user why.
|
||||
|
||||
The fix surfaces a deduped ``_emit_warning`` from ``build_turn_context`` and
|
||||
exposes ``ContextCompressor.should_compress_info`` (a ``(bool, reason)`` tuple)
|
||||
so callers can tell *why* compression was skipped while still over threshold.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from unittest.mock import patch
|
||||
|
||||
from agent.context_compressor import ContextCompressor
|
||||
from agent.turn_context import build_turn_context
|
||||
from tests.agent.test_turn_context import _FakeAgent
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Unit tests for ContextCompressor.should_compress_info
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _make_compressor(**kwargs) -> ContextCompressor:
|
||||
defaults = dict(
|
||||
model="test-model",
|
||||
threshold_percent=0.65,
|
||||
protect_first_n=2,
|
||||
protect_last_n=3,
|
||||
quiet_mode=True,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
# 96K context -> small-context floor raises threshold_percent to 0.75,
|
||||
# so threshold_tokens = 72_000. 73_000 is "over threshold".
|
||||
with patch("agent.context_compressor.get_model_context_length", return_value=96000):
|
||||
return ContextCompressor(**defaults)
|
||||
|
||||
|
||||
class TestShouldCompressInfo:
|
||||
def test_below_threshold_is_clear(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 10_000
|
||||
should, reason = comp.should_compress_info(10_000)
|
||||
assert should is False
|
||||
assert reason is None
|
||||
|
||||
def test_over_threshold_runs(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
should, reason = comp.should_compress_info(73_000)
|
||||
assert should is True
|
||||
assert reason is None
|
||||
|
||||
def test_cooldown_reports_reason(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 60
|
||||
should, reason = comp.should_compress_info(73_000)
|
||||
assert should is False
|
||||
assert reason is not None
|
||||
assert reason.startswith("cooldown:")
|
||||
|
||||
def test_cooldown_reason_has_seconds(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 42
|
||||
_should, reason = comp.should_compress_info(73_000)
|
||||
assert reason == f"cooldown:{42:.0f}"
|
||||
|
||||
def test_ineffective_reports_reason(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._ineffective_compression_count = 2
|
||||
should, reason = comp.should_compress_info(73_000)
|
||||
assert should is False
|
||||
assert reason == "ineffective"
|
||||
|
||||
def test_should_compress_bool_shim_unchanged(self):
|
||||
"""should_compress() must still return a bare bool for existing
|
||||
callers in conversation_loop.py (and/or chains)."""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 60
|
||||
result = comp.should_compress(73_000)
|
||||
assert result is False
|
||||
assert not isinstance(result, tuple)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Integration tests: build_turn_context surfaces the warning
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class _WarnAgent(_FakeAgent):
|
||||
"""_FakeAgent already covers the prologue; we just enable compression and
|
||||
record _emit_warning calls (the base class now has a MagicMock for it)."""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.compression_enabled = True
|
||||
self._warnings = []
|
||||
self._compress_calls = 0
|
||||
# Replace the MagicMock with a recorder so we can assert contents.
|
||||
self._emit_warning = lambda message: self._warnings.append(message)
|
||||
|
||||
def _compress_context(self, messages, *a, **k):
|
||||
self._compress_calls += 1
|
||||
return messages, "SYSTEM"
|
||||
|
||||
|
||||
def _build_warn_agent(compressor: ContextCompressor) -> _WarnAgent:
|
||||
agent = _WarnAgent()
|
||||
agent.context_compressor = compressor
|
||||
return agent
|
||||
|
||||
|
||||
def _run_build(agent):
|
||||
"""Run build_turn_context with the prologue-side effects stubbed."""
|
||||
with patch("agent.auxiliary_client.set_runtime_main", lambda *a, **k: None), \
|
||||
patch("agent.turn_context._should_run_preflight_estimate", return_value=True), \
|
||||
patch("agent.turn_context.estimate_request_tokens_rough", return_value=999_999):
|
||||
return build_turn_context(
|
||||
agent=agent,
|
||||
user_message="hello",
|
||||
system_message=None,
|
||||
conversation_history=None,
|
||||
task_id=None,
|
||||
stream_callback=None,
|
||||
persist_user_message=None,
|
||||
restore_or_build_system_prompt=lambda *a, **k: None,
|
||||
install_safe_stdio=lambda: None,
|
||||
sanitize_surrogates=lambda s: s,
|
||||
summarize_user_message_for_log=lambda s: s,
|
||||
set_session_context=lambda _sid: None,
|
||||
set_current_write_origin=lambda _o: None,
|
||||
ra=lambda: type("R", (), {"_set_interrupt": lambda *a, **k: None})(),
|
||||
)
|
||||
|
||||
|
||||
class TestTurnContextOverflowWarning:
|
||||
def test_warns_on_cooldown_block(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
assert len(agent._warnings) == 1
|
||||
assert "over the compression threshold" in agent._warnings[0]
|
||||
assert "blocked (cooldown:" in agent._warnings[0]
|
||||
|
||||
def test_warns_on_ineffective_block(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._ineffective_compression_count = 2
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
assert len(agent._warnings) == 1
|
||||
assert "blocked (ineffective)" in agent._warnings[0]
|
||||
|
||||
def test_no_warning_when_compression_runs(self):
|
||||
"""When compression actually runs, no overflow warning is emitted."""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000 # over threshold, no block
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
assert agent._warnings == []
|
||||
# compression was triggered instead
|
||||
assert agent._compress_calls > 0
|
||||
|
||||
def test_dedup_does_not_spam(self):
|
||||
"""Two turns with the same block kind fire the warning only once."""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
_run_build(agent) # second turn, same cooldown kind
|
||||
assert len(agent._warnings) == 1
|
||||
|
||||
def test_warning_refires_after_block_clears(self):
|
||||
"""Once the block clears, a later block of the same kind warns again."""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
assert len(agent._warnings) == 1
|
||||
# Block clears: simulate the cooldown expiring.
|
||||
comp._summary_failure_cooldown_until = 0.0
|
||||
agent._last_ctx_overflow_warn = None
|
||||
# Re-arm the same block kind.
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
_run_build(agent)
|
||||
assert len(agent._warnings) == 2
|
||||
|
||||
def test_warning_kind_switch_refires(self):
|
||||
"""Switching block kind (cooldown -> ineffective) re-warns."""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
assert len(agent._warnings) == 1
|
||||
# Now ineffective instead of cooldown.
|
||||
comp._summary_failure_cooldown_until = 0.0
|
||||
comp._ineffective_compression_count = 2
|
||||
_run_build(agent)
|
||||
assert len(agent._warnings) == 2
|
||||
assert "blocked (ineffective)" in agent._warnings[1]
|
||||
Loading…
Add table
Add a link
Reference in a new issue