hermes-agent/tests/plugins/model_providers/test_custom_profile.py
Teknium 6b81590c55
test: prune low-value tests suite-wide (wave 1) — 46,820 → 28,106 test functions
Systematic prune per AGENTS.md test policy, one pass over every major
test tree (gateway, hermes_cli, tools, agent, run_agent, plugins, cli,
cron, tui_gateway, honcho/openviking, root-level):

- DELETE: source-reading tests (read_text/getsource on prod files),
  change-detector tests (exact catalog counts, model-name snapshots,
  config version literals), mock-echo tests (assert a mock returns what
  it was told), assertion-free/trivial tests, near-duplicate
  parametrizations (boundaries + one representative kept), async/sync
  twin duplicates, cosmetic within-file variations.
- KEEP (mandatory): security/redaction/approval guards, message-role
  alternation invariants, prompt-caching/deterministic-call-id
  invariants, issue-number regression tests (deduped), E2E tests.
- 6 test files deleted outright (script-style/no-assert or fully
  redundant); conftest.py, fakes/, fixtures/ untouched.
- tests/acp/conftest.py added: autouse fixture stubs the live
  models.dev/GitHub/Copilot/Anthropic inventory fetches that ACP server
  tests performed on every session create — test_server.py 147s → 3.4s,
  and the tests are now genuinely hermetic.
- Sleep-based slowness shrunk where safe (codex_ttfb_watchdog,
  compression_concurrent_fork, etc.); no wall-clock assertion tightened.

Verification: full hermetic suite via scripts/run_tests.sh —
2439 files, 31,130 tests passed, 0 failed, 0 flaky retries, 315s wall
(baseline: 583s wall, 13,564s subprocess CPU).
2026-07-29 13:10:23 -07:00

109 lines
4.4 KiB
Python

"""Unit tests for the custom provider profile's reasoning wiring.
``provider=custom`` covers any OpenAI-compatible endpoint the user points
Hermes at — local Ollama, vLLM, llama.cpp, and hosted reasoning APIs like
GLM-5.2 on Volcengine ARK. Before #57601's salvage, ``CustomProfile`` emitted
nothing when reasoning was *enabled*, so a configured ``reasoning_effort``
was silently dropped for every custom endpoint.
These tests pin the wire-shape contract:
- disabled → extra_body.think = False
- enabled + effort → top-level reasoning_effort (native OpenAI-compat
format GLM/ARK expect), passed through verbatim
including ``max``/``xhigh``
- enabled + no effort → nothing emitted (endpoint's server default applies)
- ollama_num_ctx → extra_body.options.num_ctx, orthogonal to reasoning
"""
from __future__ import annotations
import pytest
@pytest.fixture
def custom_profile():
"""Resolve the registered custom profile via the global registry.
Importing ``model_tools`` triggers plugin discovery, which registers the
``custom`` profile. Going through ``get_provider_profile`` keeps the test
honest — if the registered class is ever downgraded to a plain
``ProviderProfile``, the assertions below collapse.
"""
import model_tools # noqa: F401
import providers
profile = providers.get_provider_profile("custom")
assert profile is not None, "custom provider profile must be registered"
return profile
class TestCustomReasoningWireShape:
"""``build_api_kwargs_extras`` produces the correct wire format."""
def test_no_reasoning_config_emits_nothing(self, custom_profile):
"""Unset reasoning → omit everything so the endpoint's default applies."""
eb, tl = custom_profile.build_api_kwargs_extras(
reasoning_config=None, model="glm-5.2"
)
assert eb == {}
assert tl == {}
def test_disabled_sends_think_false(self, custom_profile):
"""enabled=False → reasoning_effort='none' top-level + think=False.
Both fields are required: Ollama's /v1/chat/completions silently
ignores extra_body.think (only /api/chat honours it — ollama#14820)
but respects top-level reasoning_effort (#25758). think=False stays
for proxies and the native /api/chat path.
"""
eb, tl = custom_profile.build_api_kwargs_extras(
reasoning_config={"enabled": False}, model="glm-5.2"
)
assert eb == {"think": False}
assert tl == {"reasoning_effort": "none"}
def test_effort_none_sends_think_false(self, custom_profile):
"""effort='none' is the disable alias → same dual emission."""
eb, tl = custom_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": "none"}, model="glm-5.2"
)
assert eb == {"think": False}
assert tl == {"reasoning_effort": "none"}
@pytest.mark.parametrize(
"effort", ["minimal", "low", "medium", "high", "xhigh", "max"]
)
def test_enabled_effort_goes_top_level(self, custom_profile, effort):
"""enabled + effort → TOP-LEVEL reasoning_effort, passed through verbatim.
GLM-5.2/ARK and OpenAI-compatible reasoning APIs read reasoning_effort
as a top-level string, not nested in extra_body. ``max`` is GLM's
native deep-reasoning level and must survive.
"""
eb, tl = custom_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": effort}, model="glm-5.2"
)
assert tl == {"reasoning_effort": effort}
assert "reasoning_effort" not in eb
assert "think" not in eb
def test_does_not_force_think_true_on_enable(self, custom_profile):
"""We must never send think=True on enable — it's Ollama-only and
would 400 on GLM/vLLM endpoints that don't recognize it."""
eb, _ = custom_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": "high"}, model="glm-5.2"
)
assert eb.get("think") is not True
class TestCustomReasoningWithNumCtx:
"""Ollama num_ctx and reasoning are independent and compose."""
def test_num_ctx_alone(self, custom_profile):
eb, tl = custom_profile.build_api_kwargs_extras(
reasoning_config=None, ollama_num_ctx=8192, model="qwen3"
)
assert eb == {"options": {"num_ctx": 8192}}
assert tl == {}