mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
311 lines
13 KiB
Python
311 lines
13 KiB
Python
"""Tests for gateway /compress user-facing messaging."""
|
||
|
||
from datetime import datetime
|
||
from unittest.mock import AsyncMock, MagicMock, patch
|
||
|
||
import pytest
|
||
|
||
from gateway.config import GatewayConfig, Platform, PlatformConfig
|
||
from gateway.platforms.base import MessageEvent
|
||
from gateway.session import SessionEntry, SessionSource, build_session_key
|
||
|
||
|
||
def _make_source() -> SessionSource:
|
||
return SessionSource(
|
||
platform=Platform.TELEGRAM,
|
||
user_id="u1",
|
||
chat_id="c1",
|
||
user_name="tester",
|
||
chat_type="dm",
|
||
)
|
||
|
||
|
||
def _make_event(text: str = "/compress") -> MessageEvent:
|
||
return MessageEvent(text=text, source=_make_source(), message_id="m1")
|
||
|
||
|
||
def _make_history() -> list[dict[str, str]]:
|
||
return [
|
||
{"role": "user", "content": "one"},
|
||
{"role": "assistant", "content": "two"},
|
||
{"role": "user", "content": "three"},
|
||
{"role": "assistant", "content": "four"},
|
||
]
|
||
|
||
|
||
def _make_runner(history: list[dict[str, str]]):
|
||
from gateway.run import GatewayRunner
|
||
|
||
runner = object.__new__(GatewayRunner)
|
||
runner.config = GatewayConfig(
|
||
platforms={Platform.TELEGRAM: PlatformConfig(enabled=True, token="***")}
|
||
)
|
||
session_entry = SessionEntry(
|
||
session_key=build_session_key(_make_source()),
|
||
session_id="sess-1",
|
||
created_at=datetime.now(),
|
||
updated_at=datetime.now(),
|
||
platform=Platform.TELEGRAM,
|
||
chat_type="dm",
|
||
)
|
||
runner.session_store = MagicMock()
|
||
runner.session_store.get_or_create_session.return_value = session_entry
|
||
runner.session_store.load_transcript.return_value = history
|
||
runner.session_store.rewrite_transcript = MagicMock()
|
||
runner.session_store.update_session = MagicMock()
|
||
runner.session_store._save = MagicMock()
|
||
runner._session_db = None
|
||
return runner
|
||
|
||
|
||
@pytest.mark.asyncio
|
||
async def test_compress_command_works_when_auto_compaction_disabled():
|
||
"""compression.enabled: false disables *automatic* compaction only.
|
||
|
||
The gateway /compress handler has never gated on the flag — pin that
|
||
contract (every manual-compress surface must allow manual compression
|
||
regardless of the auto toggle, #64438) and the force=True cooldown
|
||
bypass that manual compression relies on."""
|
||
history = _make_history()
|
||
compressed = [
|
||
history[0],
|
||
{"role": "assistant", "content": "compressed summary"},
|
||
history[-1],
|
||
]
|
||
runner = _make_runner(history)
|
||
agent_instance = MagicMock()
|
||
agent_instance.shutdown_memory_provider = MagicMock()
|
||
agent_instance.close = MagicMock()
|
||
agent_instance._cached_system_prompt = ""
|
||
agent_instance.tools = None
|
||
agent_instance.compression_enabled = False
|
||
agent_instance.context_compressor.has_content_to_compress.return_value = True
|
||
agent_instance.session_id = "sess-1"
|
||
agent_instance._compress_context.return_value = (compressed, "")
|
||
# Explicit non-lock-skip: MagicMock getattr would return a truthy mock.
|
||
agent_instance._compression_skipped_due_to_lock = False
|
||
|
||
def _estimate(messages, **_kwargs):
|
||
return 100 if messages == history else 60
|
||
|
||
with (
|
||
patch("gateway.run._resolve_runtime_agent_kwargs", return_value={"api_key": "test-key"}),
|
||
patch("gateway.run._resolve_gateway_model", return_value="test-model"),
|
||
patch("run_agent.AIAgent", return_value=agent_instance),
|
||
patch("agent.model_metadata.estimate_request_tokens_rough", side_effect=_estimate),
|
||
):
|
||
result = await runner._handle_compress_command(_make_event())
|
||
|
||
assert "disabled" not in result.lower()
|
||
assert "Compressed:" in result
|
||
agent_instance._compress_context.assert_called_once()
|
||
assert agent_instance._compress_context.call_args.kwargs.get("force") is True
|
||
|
||
|
||
@pytest.mark.asyncio
|
||
async def test_compress_command_surfaces_aux_model_failure_even_when_recovered():
|
||
"""When the user's configured ``auxiliary.compression.model`` errors out
|
||
but compression recovers by retrying on the main model, /compress must
|
||
STILL inform the user. Silent recovery hides broken config the user
|
||
needs to fix."""
|
||
history = _make_history()
|
||
# Compressed transcript — normal successful compression, no placeholder.
|
||
compressed = [
|
||
history[0],
|
||
{"role": "assistant", "content": "summary via main model"},
|
||
history[-1],
|
||
]
|
||
runner = _make_runner(history)
|
||
agent_instance = MagicMock()
|
||
agent_instance.shutdown_memory_provider = MagicMock()
|
||
agent_instance.close = MagicMock()
|
||
agent_instance._cached_system_prompt = ""
|
||
agent_instance.tools = None
|
||
agent_instance.context_compressor.has_content_to_compress.return_value = True
|
||
# Fallback placeholder was NOT used — recovery succeeded.
|
||
agent_instance.context_compressor._last_compress_aborted = False
|
||
agent_instance.context_compressor._last_summary_fallback_used = False
|
||
agent_instance.context_compressor._last_summary_dropped_count = 0
|
||
agent_instance.context_compressor._last_summary_error = None
|
||
# But the configured aux model DID fail before the retry succeeded.
|
||
agent_instance.context_compressor._last_aux_model_failure_model = (
|
||
"gemini-3-flash-preview"
|
||
)
|
||
agent_instance.context_compressor._last_aux_model_failure_error = (
|
||
"404 model not found: gemini-3-flash-preview"
|
||
)
|
||
agent_instance.session_id = "sess-1"
|
||
agent_instance._compress_context.return_value = (compressed, "")
|
||
agent_instance._compression_skipped_due_to_lock = False
|
||
|
||
def _estimate(messages, **_kwargs):
|
||
if messages == history:
|
||
return 100
|
||
if messages == compressed:
|
||
return 60
|
||
raise AssertionError(f"unexpected transcript: {messages!r}")
|
||
|
||
with (
|
||
patch("gateway.run._resolve_runtime_agent_kwargs", return_value={"api_key": "***"}),
|
||
patch("gateway.run._resolve_gateway_model", return_value="test-model"),
|
||
patch("run_agent.AIAgent", return_value=agent_instance),
|
||
patch("agent.model_metadata.estimate_request_tokens_rough", side_effect=_estimate),
|
||
):
|
||
result = await runner._handle_compress_command(_make_event())
|
||
|
||
# Compression succeeded
|
||
assert "Compressed:" in result
|
||
# No ⚠️ warning (that's reserved for dropped-turns case)
|
||
assert "⚠️" not in result
|
||
# But there IS an info note about the broken aux model
|
||
assert "ℹ️" in result
|
||
assert "gemini-3-flash-preview" in result
|
||
assert "404" in result
|
||
assert "auxiliary.compression.model" in result
|
||
# The user's context is explicitly called out as intact
|
||
assert "intact" in result
|
||
agent_instance.shutdown_memory_provider.assert_called_once()
|
||
agent_instance.close.assert_called_once()
|
||
|
||
|
||
@pytest.mark.asyncio
|
||
async def test_compress_command_in_place_skips_destructive_rewrite():
|
||
"""In-place compaction (compression.in_place / #38763) persists via
|
||
archive_and_compact() inside _compress_context — the previous active rows
|
||
are soft-archived and the compacted set inserted. Calling
|
||
rewrite_transcript() afterwards would invoke
|
||
replace_messages(active_only=False), DELETEing the just-archived rows
|
||
(silent data loss, #61145). The handler must skip the rewrite and still
|
||
report success."""
|
||
history = _make_history()
|
||
compressed = [
|
||
history[0],
|
||
{"role": "assistant", "content": "compacted summary"},
|
||
history[-1],
|
||
]
|
||
runner = _make_runner(history)
|
||
runner._session_db = object()
|
||
session_entry = runner.session_store.get_or_create_session.return_value
|
||
runner.session_store.rewrite_transcript = MagicMock()
|
||
|
||
agent_instance = MagicMock()
|
||
agent_instance.shutdown_memory_provider = MagicMock()
|
||
agent_instance.close = MagicMock()
|
||
agent_instance._cached_system_prompt = ""
|
||
agent_instance.tools = None
|
||
agent_instance.context_compressor.has_content_to_compress.return_value = True
|
||
# In-place compaction: session_id is UNCHANGED but marked as a success.
|
||
agent_instance._last_compaction_in_place = True
|
||
agent_instance.session_id = "sess-1"
|
||
agent_instance._compress_context.return_value = (compressed, "")
|
||
agent_instance._compression_skipped_due_to_lock = False
|
||
|
||
def _estimate(messages, **_kwargs):
|
||
if messages == history:
|
||
return 100
|
||
if messages == compressed:
|
||
return 60
|
||
raise AssertionError(f"unexpected transcript: {messages!r}")
|
||
|
||
with (
|
||
patch("gateway.run._resolve_runtime_agent_kwargs", return_value={"api_key": "***"}),
|
||
patch("gateway.run._resolve_gateway_model", return_value="test-model"),
|
||
patch("run_agent.AIAgent", return_value=agent_instance),
|
||
patch("agent.model_metadata.estimate_request_tokens_rough", side_effect=_estimate),
|
||
):
|
||
result = await runner._handle_compress_command(_make_event())
|
||
|
||
assert "Compressed:" in result
|
||
# The destructive rewrite must NOT run — archive_and_compact() already
|
||
# persisted, and rewrite_transcript would wipe the archived rows.
|
||
runner.session_store.rewrite_transcript.assert_not_called()
|
||
assert session_entry.session_id == "sess-1"
|
||
agent_instance.shutdown_memory_provider.assert_called_once()
|
||
agent_instance.close.assert_called_once()
|
||
|
||
|
||
@pytest.mark.asyncio
|
||
async def test_compress_command_preserves_platform_and_gateway_session_key():
|
||
"""The temporary compression agent must carry the originating source's
|
||
platform and stable gateway session key, matching a normal gateway turn.
|
||
Without them ``_session_source_for_agent`` falls back to a default "cli"
|
||
host source, so an external context engine misattributes the retained
|
||
transcript tail and later duplicates it on resume (#50422)."""
|
||
history = _make_history()
|
||
runner = _make_runner(history)
|
||
agent_instance = MagicMock()
|
||
agent_instance.shutdown_memory_provider = MagicMock()
|
||
agent_instance.close = MagicMock()
|
||
agent_instance._cached_system_prompt = ""
|
||
agent_instance.tools = None
|
||
agent_instance.context_compressor.has_content_to_compress.return_value = True
|
||
agent_instance.session_id = "sess-1"
|
||
agent_instance._compress_context.return_value = (list(history), "")
|
||
agent_instance._compression_skipped_due_to_lock = False
|
||
|
||
with (
|
||
patch("gateway.run._resolve_runtime_agent_kwargs", return_value={"api_key": "test-key"}),
|
||
patch("gateway.run._resolve_gateway_model", return_value="test-model"),
|
||
patch("run_agent.AIAgent", return_value=agent_instance) as mock_agent,
|
||
patch("agent.model_metadata.estimate_request_tokens_rough", return_value=100),
|
||
):
|
||
await runner._handle_compress_command(_make_event())
|
||
|
||
assert mock_agent.call_count == 1
|
||
_, kwargs = mock_agent.call_args
|
||
# Platform preserved as the live turn's config key (TELEGRAM -> "telegram"),
|
||
# not the unbound "cli"/"local" fallback.
|
||
assert kwargs.get("platform") == "telegram"
|
||
# Stable gateway session key preserved, identical to a normal gateway turn.
|
||
assert kwargs.get("gateway_session_key") == runner._session_key_for_source(_make_source())
|
||
assert kwargs["gateway_session_key"]
|
||
|
||
|
||
@pytest.mark.asyncio
|
||
async def test_compress_command_passes_tool_messages_to_compressor():
|
||
"""Tool results must reach _compress_context (#3854).
|
||
|
||
Filtering the transcript to user/assistant-only starved the
|
||
compressor's tool-result pruning — tool messages are usually the bulk
|
||
of the context.
|
||
"""
|
||
history = [
|
||
{"role": "user", "content": "run it"},
|
||
{
|
||
"role": "assistant",
|
||
"content": None,
|
||
"tool_calls": [{"id": "t1", "type": "function",
|
||
"function": {"name": "x", "arguments": "{}"}}],
|
||
},
|
||
{"role": "tool", "content": "BIG RESULT " * 50, "tool_call_id": "t1"},
|
||
{"role": "assistant", "content": "done"},
|
||
{"role": "user", "content": "thanks"},
|
||
{"role": "assistant", "content": "np"},
|
||
]
|
||
runner = _make_runner(history)
|
||
agent_instance = MagicMock()
|
||
agent_instance.shutdown_memory_provider = MagicMock()
|
||
agent_instance.close = MagicMock()
|
||
agent_instance._cached_system_prompt = ""
|
||
agent_instance.tools = None
|
||
agent_instance.context_compressor.has_content_to_compress.return_value = True
|
||
agent_instance.session_id = "sess-1"
|
||
agent_instance._compress_context.return_value = (list(history), "")
|
||
|
||
with (
|
||
patch("gateway.run._resolve_runtime_agent_kwargs", return_value={"api_key": "test-key"}),
|
||
patch("gateway.run._resolve_gateway_model", return_value="test-model"),
|
||
patch("run_agent.AIAgent", return_value=agent_instance),
|
||
patch("agent.model_metadata.estimate_request_tokens_rough", return_value=100),
|
||
):
|
||
await runner._handle_compress_command(_make_event())
|
||
|
||
args, _kwargs = agent_instance._compress_context.call_args
|
||
passed = args[0]
|
||
roles = [m.get("role") for m in passed]
|
||
assert "tool" in roles, f"tool messages filtered out: {roles}"
|
||
# Assistant tool_calls stubs (content=None) must survive too, or the
|
||
# tool message would dangle without its call.
|
||
assert any(m.get("tool_calls") for m in passed), "assistant tool_calls stub dropped"
|
||
|
||
|