hermes-agent/tests/agent/test_codex_responses_adapter.py
Teknium 6b81590c55
test: prune low-value tests suite-wide (wave 1) — 46,820 → 28,106 test functions
Systematic prune per AGENTS.md test policy, one pass over every major
test tree (gateway, hermes_cli, tools, agent, run_agent, plugins, cli,
cron, tui_gateway, honcho/openviking, root-level):

- DELETE: source-reading tests (read_text/getsource on prod files),
  change-detector tests (exact catalog counts, model-name snapshots,
  config version literals), mock-echo tests (assert a mock returns what
  it was told), assertion-free/trivial tests, near-duplicate
  parametrizations (boundaries + one representative kept), async/sync
  twin duplicates, cosmetic within-file variations.
- KEEP (mandatory): security/redaction/approval guards, message-role
  alternation invariants, prompt-caching/deterministic-call-id
  invariants, issue-number regression tests (deduped), E2E tests.
- 6 test files deleted outright (script-style/no-assert or fully
  redundant); conftest.py, fakes/, fixtures/ untouched.
- tests/acp/conftest.py added: autouse fixture stubs the live
  models.dev/GitHub/Copilot/Anthropic inventory fetches that ACP server
  tests performed on every session create — test_server.py 147s → 3.4s,
  and the tests are now genuinely hermetic.
- Sleep-based slowness shrunk where safe (codex_ttfb_watchdog,
  compression_concurrent_fork, etc.); no wall-clock assertion tightened.

Verification: full hermetic suite via scripts/run_tests.sh —
2439 files, 31,130 tests passed, 0 failed, 0 flaky retries, 315s wall
(baseline: 583s wall, 13,564s subprocess CPU).
2026-07-29 13:10:23 -07:00

312 lines
10 KiB
Python

from types import SimpleNamespace
import pytest
from agent.codex_responses_adapter import (
_chat_messages_to_responses_input,
_format_responses_error,
_normalize_codex_response,
_preflight_codex_api_kwargs,
_preflight_codex_input_items,
)
def test_normalize_codex_response_treats_summary_only_reasoning_as_incomplete():
"""Summary-only reasoning keeps the continuation path for Codex backends.
Since #64434, an unrecognized issuer with ``response.status="completed"``
trusts the provider and returns ``stop`` — so this test pins the Codex
backend explicitly, where reasoning-only still means "still thinking".
"""
response = SimpleNamespace(
status="completed",
output=[
SimpleNamespace(
type="reasoning",
id="rs_tmp_789",
encrypted_content="opaque-transient",
summary=[SimpleNamespace(text="still thinking")],
)
],
)
assistant_message, finish_reason = _normalize_codex_response(
response, issuer_kind="codex_backend"
)
assert finish_reason == "incomplete"
assert assistant_message.content == ""
assert assistant_message.reasoning == "still thinking"
assert assistant_message.codex_reasoning_items is None
# ---------------------------------------------------------------------------
# Server-side built-in tool calls (xAI native web_search, code interpreter,
# etc.) come back as discrete ``*_call`` output items that xAI's
# /v1/responses surface routinely leaves at ``status="in_progress"`` even
# when the overall ``response.status == "completed"``. These must NOT mark
# the turn incomplete — otherwise grok-composer-2.5-fast research queries
# (which invoke server-side web_search) get misclassified as
# ``finish_reason="incomplete"`` and burn 3 fruitless continuation retries
# before failing with "Codex response remained incomplete after 3
# continuation attempts". Observed live against grok-composer-2.5-fast on
# SuperGrok OAuth (2026-06).
# ---------------------------------------------------------------------------
# ---------------------------------------------------------------------------
# Replayed assistant message items with an oversized server-assigned ``id``
# (Codex issues 400+ char base64 blobs) must never reach the API — the
# Responses endpoint caps input[].id at 64 chars and rejects the whole
# request with a non-retryable HTTP 400, permanently bricking the session
# (every subsequent turn replays the same bad id). Short ids (msg_...) are
# still worth keeping for prefix-cache hits, so this is a length guard, not
# a blanket strip.
# ---------------------------------------------------------------------------
_OVERSIZED_ITEM_ID = "x" * 408
_VALID_ITEM_ID = "msg_abc123"
# The codex app-server overflows the Responses 64-char call_id limit for
# MCP-routed tools, e.g. codex_mcp__hermes-tools__web_search_exec-<uuid> (#73492).
_OVERSIZED_CALL_ID = "codex_mcp__hermes-tools__web_search_exec-" + "0" * 43
def test_chat_messages_to_responses_input_clamps_oversized_call_id():
"""An oversized call_id must be clamped to <=64 chars on BOTH the
function_call and its matching function_call_output, to the same surrogate,
so the pairing survives (#73492)."""
messages = [
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"call_id": _OVERSIZED_CALL_ID,
"function": {"name": "web_search", "arguments": "{}"},
}
],
},
{
"role": "tool",
"tool_call_id": _OVERSIZED_CALL_ID,
"content": "some result",
},
]
items = _chat_messages_to_responses_input(messages)
call = next(i for i in items if i.get("type") == "function_call")
output = next(i for i in items if i.get("type") == "function_call_output")
assert len(call["call_id"]) <= 64
assert call["call_id"] != _OVERSIZED_CALL_ID
# Deterministic surrogate — the pair must still reference the same id.
assert call["call_id"] == output["call_id"]
def test_chat_messages_to_responses_input_keeps_short_call_id():
"""A call_id already within the limit passes through unchanged (#73492)."""
messages = [
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"call_id": "call_abc123",
"function": {"name": "web_search", "arguments": "{}"},
}
],
},
{
"role": "tool",
"tool_call_id": "call_abc123",
"content": "some result",
},
]
items = _chat_messages_to_responses_input(messages)
call = next(i for i in items if i.get("type") == "function_call")
output = next(i for i in items if i.get("type") == "function_call_output")
assert call["call_id"] == "call_abc123"
assert output["call_id"] == "call_abc123"
def test_preflight_codex_input_items_drops_short_id_for_github_responses():
items = _preflight_codex_input_items(
[
{
"type": "message",
"role": "assistant",
"status": "in_progress",
"content": [{"type": "output_text", "text": "pong"}],
"id": _VALID_ITEM_ID,
"phase": "final_answer",
}
],
is_github_responses=True,
)
assert "id" not in items[0]
assert items[0]["status"] == "in_progress"
assert items[0]["phase"] == "final_answer"
assert items[0]["content"] == [{"type": "output_text", "text": "pong"}]
def test_preflight_codex_api_kwargs_drops_oversized_message_id_end_to_end():
kwargs = _preflight_codex_api_kwargs(
{
"model": "gpt-5.5",
"instructions": "You are Hermes.",
"input": [
{"role": "user", "content": "ping"},
{
"type": "message",
"role": "assistant",
"status": "completed",
"content": [{"type": "output_text", "text": "pong"}],
"id": _OVERSIZED_ITEM_ID,
"phase": "final_answer",
},
],
"tools": [],
"store": False,
}
)
message_item = next(item for item in kwargs["input"] if item.get("type") == "message")
assert "id" not in message_item
# ---------------------------------------------------------------------------
# _preflight_codex_api_kwargs — built-in (provider-executed) tools must pass
# through validation. Regression guard for the xAI native web_search
# injection: the preflight validator previously rejected any tool whose
# ``type != "function"`` with "unsupported type", which would 400 every xAI
# turn once the native web_search tool is declared.
# ---------------------------------------------------------------------------
def test_preflight_passes_native_web_search_tool_through():
kwargs = {
"model": "grok-composer-2.5-fast",
"instructions": "You are helpful.",
"input": [{"role": "user", "content": [{"type": "input_text", "text": "hi"}]}],
"store": False,
"tools": [
{"type": "function", "name": "read_file", "description": "Read.",
"parameters": {"type": "object", "properties": {}}},
{"type": "web_search"},
],
}
out = _preflight_codex_api_kwargs(kwargs, allow_stream=True)
tools = out["tools"]
assert {"type": "web_search"} in tools
assert any(t.get("type") == "function" and t.get("name") == "read_file" for t in tools)
# ---------------------------------------------------------------------------
# _format_responses_error — adapted from anomalyco/opencode#28757.
# Provider failures should surface BOTH the code (rate_limit_exceeded /
# context_length_exceeded / internal_error / server_error) and the message,
# so consumers can tell rate limits apart from context-length failures and
# both apart from generic stream drops.
# ---------------------------------------------------------------------------
def test_format_responses_error_message_only():
err = {"message": "Upstream model unavailable"}
assert _format_responses_error(err, "failed") == "Upstream model unavailable"
def test_normalize_codex_response_failed_includes_code_in_error():
"""Regression: response_status == 'failed' should surface the error
code, not just the message. Used to leak a bare 'Slow down' string
that was indistinguishable from a generic stream truncation."""
# ``output`` non-empty so we don't trip the "no output items" guard
# before reaching the failed-status branch. Real failed responses
# often DO carry a partial message item alongside the error.
response = SimpleNamespace(
status="failed",
output=[
SimpleNamespace(
type="message",
role="assistant",
status="incomplete",
content=[SimpleNamespace(type="output_text", text="partial")],
),
],
error={"code": "rate_limit_exceeded", "message": "Slow down"},
)
with pytest.raises(RuntimeError, match=r"^rate_limit_exceeded: Slow down$"):
_normalize_codex_response(response)
# ---------------------------------------------------------------------------
# Reasoning-channel answer salvage (xAI grok) — grok-4.x on the xAI
# /v1/responses surface sometimes emits its final answer inside the
# reasoning item, delimited by grok's internal "<response>" tag, with no
# ``message`` output item at all. Because those reasoning items carry no
# encrypted_content, the interim message replays as nothing and every
# continuation request is byte-identical — the turn burns 3 retries and
# fails even though the answer was produced. Observed live with grok-4.20
# on xai-oauth (2026-07-13).
# ---------------------------------------------------------------------------
def _xai_reasoning_only_response(reasoning_text):
return SimpleNamespace(
status="completed",
output=[
SimpleNamespace(
type="reasoning",
id="rs_1",
encrypted_content=None,
summary=[SimpleNamespace(text=reasoning_text)],
)
],
)