mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
140 lines
4.2 KiB
Python
140 lines
4.2 KiB
Python
"""Tests for the specifier module + `hermes kanban specify` CLI surface.
|
|
|
|
The auxiliary LLM client is mocked — these tests don't hit any network or
|
|
real provider. They exercise the prompt plumbing, response parsing, DB
|
|
writes, and CLI flag surface.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json as jsonlib
|
|
from pathlib import Path
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
from hermes_cli import kanban as kanban_cli
|
|
from hermes_cli import kanban_db as kb
|
|
from hermes_cli import kanban_specify as spec
|
|
|
|
|
|
@pytest.fixture
|
|
def kanban_home(tmp_path, monkeypatch):
|
|
home = tmp_path / ".hermes"
|
|
home.mkdir()
|
|
monkeypatch.setenv("HERMES_HOME", str(home))
|
|
monkeypatch.setattr(Path, "home", lambda: tmp_path)
|
|
kb.init_db()
|
|
return home
|
|
|
|
|
|
def _fake_aux_response(content: str):
|
|
"""Build a minimal object shaped like an OpenAI chat.completions result.
|
|
|
|
The specifier only reads ``resp.choices[0].message.content``, so we
|
|
avoid importing the openai SDK and build the tree with MagicMock.
|
|
"""
|
|
resp = MagicMock()
|
|
resp.choices = [MagicMock()]
|
|
resp.choices[0].message.content = content
|
|
return resp
|
|
|
|
|
|
def _mock_client_returning(content: str):
|
|
client = MagicMock()
|
|
client.chat.completions.create = MagicMock(return_value=_fake_aux_response(content))
|
|
return client
|
|
|
|
|
|
def _patch_aux_client(content: str, *, model: str = "test-model"):
|
|
"""Patch call_llm at its source module — specify_task now routes through
|
|
it (#35566) instead of building a raw client. Returns (patcher, mock) so
|
|
callers can still assert on the call.
|
|
"""
|
|
mock_fn = MagicMock(return_value=_fake_aux_response(content))
|
|
return patch("agent.auxiliary_client.call_llm", mock_fn), mock_fn
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# JSON extraction helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# specify_task (module-level entry point)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def test_specify_task_happy_path(kanban_home):
|
|
with kb.connect() as conn:
|
|
tid = kb.create_task(conn, title="rough", triage=True)
|
|
|
|
content = jsonlib.dumps({
|
|
"title": "Refined rough",
|
|
"body": "**Goal**\nA concrete goal.",
|
|
})
|
|
p, _ = _patch_aux_client(content)
|
|
with p:
|
|
outcome = spec.specify_task(tid, author="ace")
|
|
|
|
assert outcome.ok is True
|
|
assert outcome.task_id == tid
|
|
assert outcome.new_title == "Refined rough"
|
|
|
|
with kb.connect() as conn:
|
|
task = kb.get_task(conn, tid)
|
|
# Parent-free → recompute_ready promotes to ready.
|
|
assert task.status == "ready"
|
|
assert task.title == "Refined rough"
|
|
assert "**Goal**" in (task.body or "")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# CLI wiring — argparse + _cmd_specify
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _run_cli(*argv: str) -> int:
|
|
"""Invoke the `hermes kanban …` argparse surface directly."""
|
|
root = argparse.ArgumentParser()
|
|
subp = root.add_subparsers(dest="cmd")
|
|
kanban_cli.build_parser(subp)
|
|
ns = root.parse_args(["kanban", *argv])
|
|
return kanban_cli.kanban_command(ns)
|
|
|
|
|
|
|
|
|
|
def test_cli_specify_tenant_filter(kanban_home, capsys):
|
|
with kb.connect() as conn:
|
|
outside = kb.create_task(conn, title="outside", triage=True)
|
|
inside = kb.create_task(
|
|
conn, title="inside", triage=True, tenant="proj-a",
|
|
)
|
|
|
|
content = jsonlib.dumps({"title": "spec", "body": "body"})
|
|
p, _ = _patch_aux_client(content)
|
|
with p:
|
|
rc = _run_cli("specify", "--all", "--tenant", "proj-a", "--json")
|
|
assert rc == 0
|
|
lines = [
|
|
jsonlib.loads(l)
|
|
for l in capsys.readouterr().out.strip().splitlines()
|
|
if l
|
|
]
|
|
ids = {row["task_id"] for row in lines}
|
|
assert ids == {inside}
|
|
|
|
# The outside task stays in triage.
|
|
with kb.connect() as conn:
|
|
assert kb.get_task(conn, outside).status == "triage"
|
|
# The inside task was promoted.
|
|
assert kb.get_task(conn, inside).status in {"todo", "ready"}
|
|
|
|
|