mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
95 lines
3.7 KiB
Python
95 lines
3.7 KiB
Python
"""Regression test for /model context-length display on provider-capped models.
|
|
|
|
Bug (April 2026): `/model gpt-5.5` on openai-codex (ChatGPT OAuth) showed
|
|
"Context: 1,050,000 tokens" because the display code used the raw models.dev
|
|
``ModelInfo.context_window`` (which reports the direct-OpenAI API value) instead
|
|
of the provider-aware resolver. The agent was actually running at 272K — Codex
|
|
OAuth's enforced cap — so the display was lying to the user.
|
|
|
|
Fix: ``resolve_display_context_length()`` prefers
|
|
``agent.model_metadata.get_model_context_length`` (which knows about Codex OAuth,
|
|
Copilot, Nous, etc.) and falls back to models.dev only if that returns nothing.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from unittest.mock import patch
|
|
|
|
from hermes_cli.model_switch import resolve_display_context_length
|
|
|
|
|
|
class _FakeModelInfo:
|
|
def __init__(self, ctx):
|
|
self.context_window = ctx
|
|
|
|
|
|
class TestResolveDisplayContextLength:
|
|
def test_codex_oauth_overrides_models_dev(self):
|
|
"""gpt-5.5 on openai-codex must show Codex's 272K cap, not models.dev's 1.05M."""
|
|
fake_mi = _FakeModelInfo(1_050_000) # what models.dev reports
|
|
with patch(
|
|
"agent.model_metadata.get_model_context_length",
|
|
return_value=272_000, # what Codex OAuth actually enforces
|
|
):
|
|
ctx = resolve_display_context_length(
|
|
"gpt-5.5",
|
|
"openai-codex",
|
|
base_url="https://chatgpt.com/backend-api/codex",
|
|
api_key="",
|
|
model_info=fake_mi,
|
|
)
|
|
assert ctx == 272_000, (
|
|
"Codex OAuth's 272K cap must win over models.dev's 1.05M for gpt-5.5"
|
|
)
|
|
|
|
|
|
|
|
|
|
def test_prefers_resolver_even_when_model_info_has_larger_value(self):
|
|
"""Invariant: provider-aware resolver is authoritative, even if models.dev
|
|
reports a bigger window."""
|
|
fake_mi = _FakeModelInfo(2_000_000)
|
|
with patch(
|
|
"agent.model_metadata.get_model_context_length", return_value=128_000
|
|
):
|
|
ctx = resolve_display_context_length(
|
|
"capped-model",
|
|
"capped-provider",
|
|
model_info=fake_mi,
|
|
)
|
|
assert ctx == 128_000
|
|
|
|
def test_custom_providers_override_honored(self):
|
|
"""Regression for #15779: /model switch onto a custom provider must
|
|
surface the configured per-model context_length, not the 128K/256K
|
|
fallback.
|
|
"""
|
|
custom_provs = [
|
|
{
|
|
"name": "my-custom-endpoint",
|
|
"base_url": "https://example.invalid/v1",
|
|
"models": {"gpt-5.5": {"context_length": 1_050_000}},
|
|
}
|
|
]
|
|
# Real resolver call — no mock — so the override path is exercised
|
|
# through agent.model_metadata.get_model_context_length.
|
|
from unittest.mock import patch as _p
|
|
from agent import model_metadata as _mm
|
|
with _p.object(_mm, "get_cached_context_length", return_value=None), \
|
|
_p.object(_mm, "fetch_endpoint_model_metadata", return_value={}), \
|
|
_p.object(_mm, "fetch_model_metadata", return_value={}), \
|
|
_p.object(_mm, "is_local_endpoint", return_value=False), \
|
|
_p.object(_mm, "_is_known_provider_base_url", return_value=False):
|
|
ctx = resolve_display_context_length(
|
|
"gpt-5.5",
|
|
"custom",
|
|
base_url="https://example.invalid/v1",
|
|
api_key="k",
|
|
custom_providers=custom_provs,
|
|
)
|
|
assert ctx == 1_050_000, (
|
|
"custom_providers[].models.gpt-5.5.context_length=1.05M must win "
|
|
"over probe-down fallback"
|
|
)
|
|
|
|
|
|
|