mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-30 19:09:28 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
55 lines
2.2 KiB
Python
55 lines
2.2 KiB
Python
"""Behavior tests for _clear_conversation_scope — the single conversation-
|
|
boundary funnel (#64934 follow-up).
|
|
|
|
Boundaries (/new, /resume, auto-reset, expiry finalization,
|
|
compression-exhausted reset) used to each carry a hand-copied pop-list of the
|
|
per-session dicts, and the lists drifted whenever a new dict was added
|
|
(#48031, #58403, #10702, #35809 were all "boundary X forgot dict Y" bugs).
|
|
The funnel clears every dict registered in _CONVERSATION_SCOPED_STATE plus
|
|
the boundary security state, in one call.
|
|
"""
|
|
|
|
from gateway.run import _CONVERSATION_SCOPED_STATE, GatewayRunner
|
|
|
|
KEY = "agent:main:telegram:dm:777"
|
|
OTHER = "agent:main:discord:dm:888"
|
|
|
|
|
|
def _bare_runner() -> GatewayRunner:
|
|
runner = object.__new__(GatewayRunner)
|
|
for attr in _CONVERSATION_SCOPED_STATE:
|
|
setattr(runner, attr, {KEY: object(), OTHER: object()})
|
|
# Turn-scoped state that the funnel must NOT touch.
|
|
runner._running_agents = {KEY: object()}
|
|
runner._running_agents_ts = {KEY: 1.0}
|
|
runner._session_run_generation = {KEY: 7}
|
|
return runner
|
|
|
|
|
|
def test_funnel_leaves_turn_scoped_and_generation_state_alone():
|
|
runner = _bare_runner()
|
|
runner._clear_conversation_scope(KEY, reason="test")
|
|
# Turn-scoped: owned by _release_running_agent_state / dispatch finally.
|
|
assert KEY in runner._running_agents
|
|
assert KEY in runner._running_agents_ts
|
|
# Generation counter is monotonic by design (#28686) — never reset.
|
|
assert runner._session_run_generation[KEY] == 7
|
|
|
|
|
|
def test_funnel_is_bare_runner_safe_and_empty_key_noop():
|
|
runner = object.__new__(GatewayRunner)
|
|
# No dicts initialized at all — must not raise (pitfall #17).
|
|
runner._clear_conversation_scope(KEY, reason="test")
|
|
runner._clear_conversation_scope("", reason="test")
|
|
|
|
|
|
def test_funnel_also_clears_boundary_security_state():
|
|
runner = _bare_runner()
|
|
runner._pending_approvals = {KEY: {"cmd": "rm -rf"}, OTHER: {}}
|
|
runner._update_prompt_pending = {KEY: True}
|
|
runner._pending_skills_reload_notes = {KEY: "note"}
|
|
runner._clear_conversation_scope(KEY, reason="test")
|
|
assert KEY not in runner._pending_approvals
|
|
assert OTHER in runner._pending_approvals
|
|
assert KEY not in runner._update_prompt_pending
|
|
assert KEY not in runner._pending_skills_reload_notes
|