hermes-agent/tests/gateway/test_stacked_skill_platform_disabled.py
Teknium 39975613b1
test: prune wave 2 + speed fixes — 28,106 → 19,757 test functions, suite wall 315s → 294s
Second, deeper pass over tools/gateway/hermes_cli plus first pass over
the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker,
dashboard, conformance, monitoring, secret_sources, hermes_state,
providers). Same rubric as wave 1 (AGENTS.md test policy); security,
alternation/caching invariants, issue-number regressions, and E2E kept.

Real test-quality fixes found and rooted out along the way:
- tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls
  (DEFAULT_CONFIG smart-approval leaked in) — pinned approval
  mode=manual via autouse fixture: 17.4s → 0.4s.
- test_model_switch_custom_providers.py / test_user_providers_model_switch.py
  silently probed live provider catalogs (~2s/test) — stubbed
  cached_provider_model_ids/provider_model_ids/fetch_api_models.
- test_telegram_noise_filter.py: 15-platform copy-paste matrix over
  shared gateway.run logic → 3 representative platforms (55s → 3.9s).
- test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on
  MagicMock agents — interrupt.side_effect now clears _running_agents
  (22s → 1.0s).
- test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x
  (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps
  patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait
  5s → 0.5s.
- test_telegram_init_deadline.py: loop-block margin restored to 1.0s
  with rationale comment — the watchdog-dump assertion needs the loop
  blocked well past deadline+grace under parallel load (flaked once in
  the 40-worker verification run at a 0.2s margin).

Verification: full hermetic suite via scripts/run_tests.sh —
2,438 files, 21,718 tests passed, 0 failed, 293.9s wall.
Suite totals vs original baseline: 46,820 → 19,757 test functions
(−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
2026-07-29 13:39:40 -07:00

137 lines
4.8 KiB
Python

"""Regression test for stacked slash-skill invocations bypassing the
per-platform ``skills.platform_disabled`` gate.
``/skill-a /skill-b do XYZ`` loads every leading skill (up to 5), not just
the first (``agent.skill_commands.split_stacked_skill_commands`` /
``build_stacked_skill_invocation_message``). ``gateway.run.GatewayRunner.
_handle_message`` already re-checks the FIRST skill against the
per-platform disabled list before dispatch (``get_skill_commands()`` only
applies the *global* disabled list at scan time), but did not extend that
same check to the additional stacked skills — a skill an operator disabled
for a given platform still had its full SKILL.md content injected into the
agent's context for that turn if it was stacked behind an allowed one.
"""
from datetime import datetime
from types import SimpleNamespace
from unittest.mock import AsyncMock, MagicMock
import pytest
from gateway.config import GatewayConfig, Platform, PlatformConfig
from gateway.platforms.base import MessageEvent
from gateway.session import SessionEntry, SessionSource, build_session_key
def _make_source() -> SessionSource:
return SessionSource(
platform=Platform.TELEGRAM,
user_id="u1",
chat_id="c1",
user_name="tester",
chat_type="dm",
)
def _make_event(text: str) -> MessageEvent:
return MessageEvent(text=text, source=_make_source(), message_id="m1")
def _make_runner():
from gateway.run import GatewayRunner
runner = object.__new__(GatewayRunner)
runner.config = GatewayConfig(
platforms={Platform.TELEGRAM: PlatformConfig(enabled=True, token="***")}
)
adapter = MagicMock()
adapter.send = AsyncMock()
runner.adapters = {Platform.TELEGRAM: adapter}
runner._voice_mode = {}
runner.hooks = SimpleNamespace(
emit=AsyncMock(),
emit_collect=AsyncMock(return_value=[]),
loaded_hooks=False,
)
session_entry = SessionEntry(
session_key=build_session_key(_make_source()),
session_id="sess-1",
created_at=datetime.now(),
updated_at=datetime.now(),
platform=Platform.TELEGRAM,
chat_type="dm",
)
runner.session_store = MagicMock()
runner.session_store.get_or_create_session.return_value = session_entry
runner.session_store.load_transcript.return_value = []
runner.session_store.has_any_sessions.return_value = True
runner.session_store.append_to_transcript = MagicMock()
runner.session_store.rewrite_transcript = MagicMock()
runner.session_store.update_session = MagicMock()
runner._running_agents = {}
runner._pending_messages = {}
runner._pending_approvals = {}
runner._session_db = None
runner._reasoning_config = None
runner._provider_routing = {}
runner._fallback_model = None
runner._show_reasoning = False
runner._is_user_authorized = lambda _source: True
runner._set_session_env = lambda _context: None
runner._should_send_voice_reply = lambda *_args, **_kwargs: False
from gateway.run import GatewayRunner as _GR
runner._session_key_for_source = _GR._session_key_for_source.__get__(runner, _GR)
return runner
def _make_skill(skills_dir, name, body="content"):
sd = skills_dir / name
sd.mkdir(parents=True, exist_ok=True)
(sd / "SKILL.md").write_text(
f"---\nname: {name}\ndescription: desc {name}\n---\n\n# {name}\n\n{body}\n"
)
@pytest.fixture
def skills_env(tmp_path, monkeypatch):
skills_dir = tmp_path / "skills"
skills_dir.mkdir()
import tools.skills_tool as skills_tool_module
monkeypatch.setattr(skills_tool_module, "SKILLS_DIR", skills_dir)
import agent.skill_commands as skill_commands_mod
skill_commands_mod._skill_commands = {}
skill_commands_mod._skill_commands_platform = None
return skills_dir
@pytest.mark.asyncio
async def test_stacked_second_skill_disabled_for_platform_is_blocked(monkeypatch, skills_env):
"""The whole stacked invocation is rejected when a NON-leading stacked
skill is disabled for the message's platform — it must not silently load
that skill's content just because only the first skill was checked."""
import gateway.run as gateway_run
import agent.skill_utils as skill_utils_mod
_make_skill(skills_env, "allowed-skill")
_make_skill(skills_env, "disabled-skill")
monkeypatch.setattr(
skill_utils_mod,
"get_disabled_skill_names",
lambda platform=None: {"disabled-skill"} if platform == "telegram" else set(),
)
monkeypatch.setattr(
gateway_run, "_resolve_runtime_agent_kwargs", lambda: {"api_key": "***"}
)
runner = _make_runner()
result = await runner._handle_message(
_make_event("/allowed-skill /disabled-skill do something")
)
assert result is not None
assert "disabled-skill" in result
assert "disabled for telegram" in result