mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
Systematic prune per AGENTS.md test policy, one pass over every major test tree (gateway, hermes_cli, tools, agent, run_agent, plugins, cli, cron, tui_gateway, honcho/openviking, root-level): - DELETE: source-reading tests (read_text/getsource on prod files), change-detector tests (exact catalog counts, model-name snapshots, config version literals), mock-echo tests (assert a mock returns what it was told), assertion-free/trivial tests, near-duplicate parametrizations (boundaries + one representative kept), async/sync twin duplicates, cosmetic within-file variations. - KEEP (mandatory): security/redaction/approval guards, message-role alternation invariants, prompt-caching/deterministic-call-id invariants, issue-number regression tests (deduped), E2E tests. - 6 test files deleted outright (script-style/no-assert or fully redundant); conftest.py, fakes/, fixtures/ untouched. - tests/acp/conftest.py added: autouse fixture stubs the live models.dev/GitHub/Copilot/Anthropic inventory fetches that ACP server tests performed on every session create — test_server.py 147s → 3.4s, and the tests are now genuinely hermetic. - Sleep-based slowness shrunk where safe (codex_ttfb_watchdog, compression_concurrent_fork, etc.); no wall-clock assertion tightened. Verification: full hermetic suite via scripts/run_tests.sh — 2439 files, 31,130 tests passed, 0 failed, 0 flaky retries, 315s wall (baseline: 583s wall, 13,564s subprocess CPU).
69 lines
2.3 KiB
Python
69 lines
2.3 KiB
Python
"""Tests for custom-provider model matching (extra_body / service_tier drop bug).
|
|
|
|
July 2026 incident: a custom provider with a multi-model catalog
|
|
(``models: {gpt-5.5: {}, gpt-5.6-terra: {}, ...}``) and a ``model``/
|
|
``default_model`` differing from the session model failed
|
|
``_custom_provider_model_matches``, so ``extra_body: {service_tier: flex}``
|
|
was silently dropped — every request billed at standard tier (~2.3x).
|
|
"""
|
|
|
|
from agent.agent_init import (
|
|
_custom_provider_extra_body_for_agent,
|
|
_custom_provider_model_matches,
|
|
)
|
|
|
|
BASE = "https://api.openai.com/v1"
|
|
|
|
|
|
def _entry(**over):
|
|
e = {
|
|
"name": "openai",
|
|
"base_url": BASE,
|
|
"extra_body": {"service_tier": "flex"},
|
|
}
|
|
e.update(over)
|
|
return e
|
|
|
|
|
|
class TestModelMatches:
|
|
|
|
|
|
def test_catalog_miss_falls_back_to_model_field(self):
|
|
e = _entry(model="gpt-5.5", models={"gpt-5.5": {}})
|
|
assert _custom_provider_model_matches("gpt-5.5", e)
|
|
assert not _custom_provider_model_matches("gpt-4o", e)
|
|
|
|
|
|
def test_catalog_case_insensitive(self):
|
|
e = _entry(models={"GPT-5.6-Terra": {}})
|
|
assert _custom_provider_model_matches("gpt-5.6-terra", e)
|
|
|
|
|
|
class TestExtraBodyResolution:
|
|
def test_multi_model_provider_yields_extra_body(self):
|
|
# The exact sweeper-profile shape that failed in production.
|
|
entry = _entry(
|
|
model="gpt-5.5",
|
|
models={"gpt-5.5": {}, "gpt-5.6-sol": {}, "gpt-5.6-terra": {}},
|
|
)
|
|
got = _custom_provider_extra_body_for_agent(
|
|
provider="custom", model="gpt-5.6-terra",
|
|
base_url=BASE, custom_providers=[entry],
|
|
)
|
|
assert got == {"service_tier": "flex"}
|
|
|
|
def test_non_catalog_model_gets_no_override(self):
|
|
entry = _entry(model="gpt-5.5", models={"gpt-5.5": {}})
|
|
got = _custom_provider_extra_body_for_agent(
|
|
provider="custom", model="gpt-4o",
|
|
base_url=BASE, custom_providers=[entry],
|
|
)
|
|
assert got is None
|
|
|
|
def test_non_custom_provider_unaffected(self):
|
|
entry = _entry(models={"gpt-5.6-terra": {}})
|
|
got = _custom_provider_extra_body_for_agent(
|
|
provider="openrouter", model="gpt-5.6-terra",
|
|
base_url=BASE, custom_providers=[entry],
|
|
)
|
|
assert got is None
|