mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-30 19:09:28 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
90 lines
3.1 KiB
Python
90 lines
3.1 KiB
Python
"""End-to-end credential isolation proof for multiplex mode (Workstream A).
|
|
|
|
These exercise the REAL resolution path (runtime_provider, secret scope, MCP
|
|
interpolation) rather than mocking it, proving the property that matters: two
|
|
profiles with different keys never see each other's, and an unscoped read in
|
|
multiplex mode fails closed instead of leaking.
|
|
"""
|
|
import pytest
|
|
|
|
from pathlib import Path
|
|
|
|
from agent import secret_scope as ss
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _reset(monkeypatch):
|
|
ss.set_multiplex_active(False)
|
|
yield
|
|
ss.set_multiplex_active(False)
|
|
|
|
|
|
class TestRuntimeProviderUsesScope:
|
|
"""hermes_cli.runtime_provider._getenv resolves through the secret scope."""
|
|
|
|
|
|
def test_getenv_two_profiles_isolated(self, monkeypatch):
|
|
from hermes_cli.runtime_provider import _getenv
|
|
ss.set_multiplex_active(True)
|
|
|
|
tok_a = ss.set_secret_scope({"OPENAI_API_KEY": "sk-A"})
|
|
try:
|
|
assert _getenv("OPENAI_API_KEY") == "sk-A"
|
|
finally:
|
|
ss.reset_secret_scope(tok_a)
|
|
|
|
tok_b = ss.set_secret_scope({"OPENAI_API_KEY": "sk-B"})
|
|
try:
|
|
assert _getenv("OPENAI_API_KEY") == "sk-B"
|
|
finally:
|
|
ss.reset_secret_scope(tok_b)
|
|
|
|
|
|
class TestMcpInterpolationUsesScope:
|
|
"""MCP config ${VAR} interpolation resolves through the secret scope."""
|
|
|
|
def test_interpolation_reads_scope(self, monkeypatch):
|
|
from tools.mcp_tool import _interpolate_env_vars
|
|
monkeypatch.setenv("MY_MCP_TOKEN", "global-token")
|
|
ss.set_multiplex_active(True)
|
|
tok = ss.set_secret_scope({"MY_MCP_TOKEN": "profile-token"})
|
|
try:
|
|
cfg = {"env": {"TOKEN": "${MY_MCP_TOKEN}"}}
|
|
assert _interpolate_env_vars(cfg) == {"env": {"TOKEN": "profile-token"}}
|
|
finally:
|
|
ss.reset_secret_scope(tok)
|
|
|
|
|
|
class TestProfilePathResolutionUnderMultiplexScope:
|
|
"""Profile-scoped paths must follow the per-turn _profile_runtime_scope.
|
|
|
|
The multiplexed gateway (gateway.multiplex_profiles) serves every profile
|
|
from ONE process, scoping each inbound turn with _profile_runtime_scope —
|
|
the same in-process-many-profiles topology as the desktop tui_gateway. The
|
|
profile-isolation fixes (per-call path resolution + thread context
|
|
propagation) must therefore hold under THIS scope too, not just desktop.
|
|
This is the regression guard proving reachability is not desktop-only.
|
|
"""
|
|
|
|
def _profiles(self, tmp_path):
|
|
prof_a = tmp_path / "profA"
|
|
prof_b = tmp_path / "profB"
|
|
for p in (prof_a, prof_b):
|
|
(p / "skills").mkdir(parents=True, exist_ok=True)
|
|
(p / "state").mkdir(parents=True, exist_ok=True)
|
|
return prof_a, prof_b
|
|
|
|
def test_skills_dir_follows_multiplex_scope(self, tmp_path):
|
|
from gateway.run import _profile_runtime_scope
|
|
import tools.skills_hub as sh
|
|
|
|
prof_a, prof_b = self._profiles(tmp_path)
|
|
with _profile_runtime_scope(prof_a):
|
|
a_seen = Path(sh.SKILLS_DIR)
|
|
with _profile_runtime_scope(prof_b):
|
|
b_seen = Path(sh.SKILLS_DIR)
|
|
|
|
assert a_seen == prof_a / "skills"
|
|
assert b_seen == prof_b / "skills"
|
|
|
|
|