mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-30 19:09:28 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
263 lines
10 KiB
Python
263 lines
10 KiB
Python
"""Live integration tests for file operations and terminal tools.
|
|
|
|
These tests run REAL commands through the LocalEnvironment -- no mocks.
|
|
They verify that shell noise is properly filtered, commands actually work,
|
|
and the tool outputs are EXACTLY what the agent would see.
|
|
|
|
Every test with output validates against a known-good value AND
|
|
asserts zero contamination from shell noise via _assert_clean().
|
|
"""
|
|
|
|
import pytest
|
|
|
|
|
|
import os
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
|
|
|
|
from tools.environments.local import LocalEnvironment
|
|
from tools.file_operations import ShellFileOperations
|
|
|
|
|
|
# ── Shared noise detection ───────────────────────────────────────────────
|
|
# Known shell noise patterns that should never appear in command output.
|
|
|
|
_ALL_NOISE_PATTERNS = [
|
|
"bash: cannot set terminal process group",
|
|
"bash: no job control in this shell",
|
|
"no job control in this shell",
|
|
"cannot set terminal process group",
|
|
"tcsetattr: Inappropriate ioctl for device",
|
|
"bash: ",
|
|
"Inappropriate ioctl",
|
|
"Auto-suggestions:",
|
|
]
|
|
|
|
|
|
def _assert_clean(text: str, context: str = "output"):
|
|
"""Assert text contains zero shell noise contamination."""
|
|
if not text:
|
|
return
|
|
for noise in _ALL_NOISE_PATTERNS:
|
|
assert noise not in text, (
|
|
f"Shell noise leaked into {context}: found {noise!r} in:\n"
|
|
f"{text[:500]}"
|
|
)
|
|
|
|
|
|
# ── Fixtures ─────────────────────────────────────────────────────────────
|
|
|
|
# Deterministic file content used across tests. Every byte is known,
|
|
# so any unexpected text in results is immediately caught.
|
|
SIMPLE_CONTENT = "alpha\nbravo\ncharlie\n"
|
|
NUMBERED_CONTENT = "\n".join(f"LINE_{i:04d}" for i in range(1, 51)) + "\n"
|
|
SPECIAL_CONTENT = "single 'quotes' and \"doubles\" and $VARS and `backticks` and \\backslash\n"
|
|
MULTIFILE_A = "def func_alpha():\n return 42\n"
|
|
MULTIFILE_B = "def func_bravo():\n return 99\n"
|
|
MULTIFILE_C = "nothing relevant here\n"
|
|
|
|
|
|
@pytest.fixture
|
|
def env(tmp_path):
|
|
"""A real LocalEnvironment rooted in a temp directory."""
|
|
return LocalEnvironment(cwd=str(tmp_path), timeout=15)
|
|
|
|
|
|
@pytest.fixture
|
|
def ops(env, tmp_path):
|
|
"""ShellFileOperations wired to the real local environment."""
|
|
return ShellFileOperations(env, cwd=str(tmp_path))
|
|
|
|
|
|
@pytest.fixture
|
|
def populated_dir(tmp_path):
|
|
"""A temp directory with known files for search/read tests."""
|
|
(tmp_path / "alpha.py").write_text(MULTIFILE_A)
|
|
(tmp_path / "bravo.py").write_text(MULTIFILE_B)
|
|
(tmp_path / "notes.txt").write_text(MULTIFILE_C)
|
|
(tmp_path / "data.csv").write_text("col1,col2\n1,2\n3,4\n")
|
|
return tmp_path
|
|
|
|
|
|
# ── LocalEnvironment.execute() ───────────────────────────────────────────
|
|
|
|
class TestLocalEnvironmentExecute:
|
|
def test_echo_exact_output(self, env):
|
|
result = env.execute("echo DETERMINISTIC_OUTPUT_12345")
|
|
assert result["returncode"] == 0
|
|
assert result["output"].strip() == "DETERMINISTIC_OUTPUT_12345"
|
|
_assert_clean(result["output"])
|
|
|
|
def test_printf_no_trailing_newline(self, env):
|
|
result = env.execute("printf 'exact'")
|
|
assert result["returncode"] == 0
|
|
assert result["output"] == "exact"
|
|
_assert_clean(result["output"])
|
|
|
|
|
|
def test_cat_deterministic_content(self, env, tmp_path):
|
|
f = tmp_path / "det.txt"
|
|
f.write_text(SIMPLE_CONTENT)
|
|
result = env.execute(f"cat {f}")
|
|
assert result["returncode"] == 0
|
|
assert result["output"] == SIMPLE_CONTENT
|
|
_assert_clean(result["output"])
|
|
|
|
|
|
# ── _has_command ─────────────────────────────────────────────────────────
|
|
|
|
class TestHasCommand:
|
|
def test_finds_echo(self, ops):
|
|
assert ops._has_command("echo") is True
|
|
|
|
|
|
def test_missing_command(self, ops):
|
|
assert ops._has_command("nonexistent_tool_xyz_abc_999") is False
|
|
|
|
def test_rg_or_grep_available(self, ops):
|
|
assert ops._has_command("rg") or ops._has_command("grep"), \
|
|
"Neither rg nor grep found -- search_files will break"
|
|
|
|
|
|
# ── read_file ────────────────────────────────────────────────────────────
|
|
|
|
class TestReadFile:
|
|
def test_exact_content(self, ops, tmp_path):
|
|
f = tmp_path / "exact.txt"
|
|
f.write_text(SIMPLE_CONTENT)
|
|
result = ops.read_file(str(f))
|
|
assert result.error is None
|
|
# Content has line numbers prepended, check the actual text is there
|
|
assert "alpha" in result.content
|
|
assert "bravo" in result.content
|
|
assert "charlie" in result.content
|
|
assert result.total_lines == 3
|
|
_assert_clean(result.content)
|
|
|
|
|
|
def test_no_noise_in_content(self, ops, tmp_path):
|
|
f = tmp_path / "noise_check.txt"
|
|
f.write_text("ONLY_THIS_CONTENT\n")
|
|
result = ops.read_file(str(f))
|
|
assert result.error is None
|
|
_assert_clean(result.content)
|
|
|
|
|
|
# ── write_file ───────────────────────────────────────────────────────────
|
|
|
|
class TestWriteFile:
|
|
def test_write_and_verify(self, ops, tmp_path):
|
|
path = str(tmp_path / "written.txt")
|
|
result = ops.write_file(path, SIMPLE_CONTENT)
|
|
assert result.error is None
|
|
assert result.bytes_written == len(SIMPLE_CONTENT.encode())
|
|
assert Path(path).read_text() == SIMPLE_CONTENT
|
|
|
|
|
|
def test_roundtrip_read_write(self, ops, tmp_path):
|
|
"""Write -> read back -> verify exact match."""
|
|
path = str(tmp_path / "roundtrip.txt")
|
|
ops.write_file(path, SIMPLE_CONTENT)
|
|
result = ops.read_file(path)
|
|
assert result.error is None
|
|
assert "alpha" in result.content
|
|
assert "charlie" in result.content
|
|
_assert_clean(result.content)
|
|
|
|
|
|
# ── patch_replace ────────────────────────────────────────────────────────
|
|
|
|
class TestPatchReplace:
|
|
def test_exact_replacement(self, ops, tmp_path):
|
|
path = str(tmp_path / "patch.txt")
|
|
Path(path).write_text("hello world\n")
|
|
result = ops.patch_replace(path, "world", "earth")
|
|
assert result.error is None
|
|
assert Path(path).read_text() == "hello earth\n"
|
|
|
|
|
|
def test_multiline_patch(self, ops, tmp_path):
|
|
path = str(tmp_path / "multi.txt")
|
|
Path(path).write_text("line1\nline2\nline3\n")
|
|
result = ops.patch_replace(path, "line2", "REPLACED")
|
|
assert result.error is None
|
|
assert Path(path).read_text() == "line1\nREPLACED\nline3\n"
|
|
|
|
|
|
# ── search ───────────────────────────────────────────────────────────────
|
|
|
|
class TestSearch:
|
|
def test_content_search_finds_exact_match(self, ops, populated_dir):
|
|
result = ops.search("func_alpha", str(populated_dir), target="content")
|
|
assert result.error is None
|
|
assert result.total_count >= 1
|
|
assert any("func_alpha" in m.content for m in result.matches)
|
|
for m in result.matches:
|
|
_assert_clean(m.content)
|
|
_assert_clean(m.path)
|
|
|
|
|
|
def test_search_output_has_zero_noise(self, ops, populated_dir):
|
|
"""Dedicated noise check: search must return only real content."""
|
|
result = ops.search("func", str(populated_dir), target="content")
|
|
assert result.error is None
|
|
for m in result.matches:
|
|
_assert_clean(m.content)
|
|
_assert_clean(m.path)
|
|
|
|
|
|
# ── _expand_path ─────────────────────────────────────────────────────────
|
|
|
|
class TestExpandPath:
|
|
def test_tilde_exact(self, ops):
|
|
result = ops._expand_path("~/test.txt")
|
|
expected = f"{Path.home()}/test.txt"
|
|
assert result == expected
|
|
_assert_clean(result)
|
|
|
|
|
|
def test_tilde_injection_blocked(self, ops):
|
|
"""Paths like ~; rm -rf / must NOT execute shell commands."""
|
|
malicious = "~; echo PWNED > /tmp/_hermes_injection_test"
|
|
result = ops._expand_path(malicious)
|
|
# The invalid username (contains ";") should prevent shell expansion.
|
|
# The path should be returned as-is (no expansion).
|
|
assert result == malicious
|
|
# Verify the injected command did NOT execute
|
|
assert not os.path.exists("/tmp/_hermes_injection_test")
|
|
|
|
def test_tilde_username_with_subpath(self, ops):
|
|
"""~root/file.txt should attempt expansion (valid username)."""
|
|
result = ops._expand_path("~root/file.txt")
|
|
# On most systems ~root expands to /root
|
|
if result != "~root/file.txt":
|
|
assert result.endswith("/file.txt")
|
|
assert "~" not in result
|
|
|
|
|
|
# ── Terminal output cleanliness ──────────────────────────────────────────
|
|
|
|
class TestTerminalOutputCleanliness:
|
|
"""Every command the agent might run must produce noise-free output."""
|
|
|
|
def test_echo(self, env):
|
|
result = env.execute("echo CLEAN_TEST")
|
|
assert result["output"].strip() == "CLEAN_TEST"
|
|
_assert_clean(result["output"])
|
|
|
|
def test_cat(self, env, tmp_path):
|
|
f = tmp_path / "cat_test.txt"
|
|
f.write_text("CAT_CONTENT_EXACT\n")
|
|
result = env.execute(f"cat {f}")
|
|
assert result["output"] == "CAT_CONTENT_EXACT\n"
|
|
_assert_clean(result["output"])
|
|
|
|
|
|
def test_command_v_detection(self, env):
|
|
"""This is how _has_command works -- must return clean 'yes'."""
|
|
result = env.execute("command -v cat >/dev/null 2>&1 && echo 'yes'")
|
|
assert result["output"].strip() == "yes"
|
|
_assert_clean(result["output"])
|