mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-24 16:54:43 +00:00
test: cover gemini-native max_tokens forwarding in _build_call_kwargs
Requested in review: builder-level assertions that the gemini-native branch forwards max_tokens (provider names and the native generativelanguage.googleapis.com base_url, max_tokens=600), plus a control showing gemini models on OpenAI-compatible endpoints — including Gemini's own /openai compatibility endpoint — keep the existing omission behavior (#34530). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
4ee74fa5df
commit
ead9d7b256
1 changed files with 49 additions and 0 deletions
|
|
@ -775,6 +775,55 @@ class TestBuildCallKwargsMaxTokens:
|
|||
)
|
||||
assert "max_tokens" not in kw3
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"provider,model,base_url",
|
||||
[
|
||||
("gemini", "gemini-2.5-pro", None),
|
||||
("google", "gemini-2.5-flash", None),
|
||||
(
|
||||
"custom",
|
||||
"gemini-2.5-pro",
|
||||
"https://generativelanguage.googleapis.com/v1beta",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_keeps_max_tokens_for_gemini_native(self, provider, model, base_url):
|
||||
# Native generateContent maps max_tokens → maxOutputTokens; when it is
|
||||
# omitted Gemini applies a fixed 65,535-token ceiling, which silently
|
||||
# turned MoA's reference_max_tokens into a no-op for gemini advisors.
|
||||
from agent.auxiliary_client import _build_call_kwargs
|
||||
|
||||
kwargs = _build_call_kwargs(
|
||||
provider=provider,
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=600,
|
||||
base_url=base_url,
|
||||
)
|
||||
assert kwargs["max_tokens"] == 600
|
||||
assert "max_completion_tokens" not in kwargs
|
||||
|
||||
def test_omits_max_tokens_for_gemini_model_on_openai_compatible_endpoint(self):
|
||||
# Control: the gemini branch keys on provider/base_url, never the model
|
||||
# name. A gemini model served through an OpenAI-compatible endpoint
|
||||
# keeps the default omission behavior (#34530), including Gemini's own
|
||||
# /openai compatibility endpoint.
|
||||
from agent.auxiliary_client import _build_call_kwargs
|
||||
|
||||
for provider, base_url in [
|
||||
("openrouter", "https://openrouter.ai/api/v1"),
|
||||
("custom", "https://generativelanguage.googleapis.com/v1beta/openai"),
|
||||
]:
|
||||
kwargs = _build_call_kwargs(
|
||||
provider=provider,
|
||||
model="google/gemini-2.5-pro",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=600,
|
||||
base_url=base_url,
|
||||
)
|
||||
assert "max_tokens" not in kwargs
|
||||
assert "max_completion_tokens" not in kwargs
|
||||
|
||||
|
||||
class TestNousTagsScoping:
|
||||
def test_tags_injected_when_provider_is_nous(self, monkeypatch):
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue