From bcd7e2ce8999a48e887b50a1a37b6b7812eaa1d7 Mon Sep 17 00:00:00 2001 From: Ahmett101 <297889955+Ahmett101@users.noreply.github.com> Date: Tue, 14 Jul 2026 21:54:24 +0300 Subject: [PATCH] fix(agent): add finite hard ceiling on openai-codex request time (#64507) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A large Codex request (estimated context >= 10k tokens) disables the no-byte TTFB watchdog on purpose, and openai_codex_stale_timeout_floor *raises* the stale timeout (up to 1200s at >100k tokens) so healthy gateway-scale payloads aren't aborted mid-prefill. When the backend genuinely stalls — no first byte AND no events, exactly the #64507 symptom — the request is only reclaimed at that high stale floor, so the session can hang for 13+ minutes with an idle slash_worker and no ended_at/end_reason while Desktop still shows it as active. Add a flat, finite hard ceiling on total openai-codex request time that always applies (min() of the computed stale timeout and the ceiling) regardless of the TTFB-disable / stale-floor interaction. A stalled large request is now killed at the ceiling and the retry loop / visible failure path takes over instead of hanging indefinitely. Tunable via HERMES_CODEX_HARD_TIMEOUT_SECONDS (default 600s; 0 disables the ceiling to restore pre-fix behavior). Closes #64507 --- agent/chat_completion_helpers.py | 22 ++++ tests/agent/test_codex_ttfb_watchdog.py | 140 ++++++++++++++++++++++++ 2 files changed, 162 insertions(+) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 6615dace60a8..561c2ce3726d 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -504,6 +504,28 @@ def interruptible_api_call(agent, api_kwargs: dict): if _codex_floor: _stale_timeout = max(_stale_timeout, _codex_floor) + # ── Codex absolute hard ceiling (#64507) ────────────────────────── + # For large Codex requests the no-byte TTFB watchdog is intentionally + # disabled (so legitimate backend admission / prefill isn't killed), and + # ``openai_codex_stale_timeout_floor`` *raises* the stale timeout (up to + # 1200s at >100k tokens) so healthy gateway-scale payloads aren't aborted. + # That combination means a request that is genuinely stalled at the + # backend — no first byte AND no events, exactly the issue-64507 symptom — + # is only reclaimed at the (high) stale floor, so a session can hang for + # many minutes with an idle worker and no ended_at. Add a flat, finite + # hard ceiling on total request time that ALWAYS applies to openai-codex + # requests regardless of the TTFB/stale interaction, so a stalled large + # request is recovered (retry loop / visible failure) instead of hanging + # indefinitely. Tunable via HERMES_CODEX_HARD_TIMEOUT_SECONDS (set to 0 to + # disable the ceiling entirely; that restores the pre-fix behavior). + _codex_hard_timeout = _env_float("HERMES_CODEX_HARD_TIMEOUT_SECONDS", 600.0) + if ( + _codex_watchdog_enabled + and _openai_codex_backend + and _codex_hard_timeout > 0 + ): + _stale_timeout = min(_stale_timeout, _codex_hard_timeout) + if _est_tokens_for_codex_watchdog > 100_000: _codex_idle_timeout_default = 180.0 elif _est_tokens_for_codex_watchdog > 50_000: diff --git a/tests/agent/test_codex_ttfb_watchdog.py b/tests/agent/test_codex_ttfb_watchdog.py index 283a6a8e9a8d..6a647164f93d 100644 --- a/tests/agent/test_codex_ttfb_watchdog.py +++ b/tests/agent/test_codex_ttfb_watchdog.py @@ -462,3 +462,143 @@ def test_large_codex_request_strict_ttfb_env_still_reconnects(tmp_path, monkeypa assert "codex_ttfb_kill" in closes finally: stop["flag"] = True + + +def test_large_codex_request_hard_ceiling_reclaims_silent_stall(tmp_path, monkeypatch): + """#64507 regression: a large Codex request (TTFB watchdog disabled by the + size gate, stale floor *raised*) that never emits a single byte must still + be reclaimed at a finite hard ceiling — not hang for 13+ minutes while the + worker stays idle and the session shows as active. + + Uses the real default TTFB threshold (120s) and asserts the request dies at + the hard ceiling regardless of the size-based TTFB disable. + """ + from agent import chat_completion_helpers as h + + agent = _make_codex_agent(tmp_path, monkeypatch) + # Real default TTFB threshold (no HERMES_CODEX_TTFB_* override) → for a + # >10k-token request the no-byte TTFB watchdog is auto-disabled. + monkeypatch.setenv("HERMES_CODEX_HARD_TIMEOUT_SECONDS", "3") + + closes: list = [] + dummy_client = SimpleNamespace() + monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client) + monkeypatch.setattr( + agent, "_abort_request_openai_client", + lambda c, reason=None: closes.append(reason), + ) + monkeypatch.setattr( + agent, "_close_request_openai_client", + lambda c, reason=None: closes.append(reason), + ) + + stop = {"flag": False} + + def fake_hang(api_kwargs, client=None, on_first_delta=None): + # No event marker AND no event ever: the exact issue-64507 stall. + deadline = time.time() + 120 + while time.time() < deadline and not stop["flag"] and not agent._interrupt_requested: + time.sleep(0.02) + raise RuntimeError("connection closed") + + monkeypatch.setattr(agent, "_run_codex_stream", fake_hang) + + large_input = "x" * 44_000 # ~11k estimated tokens → TTFB disabled, stale raised + t0 = time.time() + try: + with pytest.raises(TimeoutError) as excinfo: + h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": large_input}) + elapsed = time.time() - t0 + # Must die at the hard ceiling (3s), nowhere near the raised stale floor. + assert elapsed < 30, f"hard ceiling took {elapsed:.1f}s — stall not reclaimed" + assert "stale_call_kill" in closes, f"stale kill expected, got {closes}" + assert "timed out after" in str(excinfo.value) + assert "with no response" in str(excinfo.value) + finally: + stop["flag"] = True + + +def test_large_codex_request_hard_ceiling_disabled_restores_legacy(tmp_path, monkeypatch): + """Setting HERMES_CODEX_HARD_TIMEOUT_SECONDS=0 disables the ceiling entirely, + restoring the pre-#64507 behavior (request waits out the raised stale floor + instead of being capped). Keeps the knob for operators who must. + """ + from agent import chat_completion_helpers as h + + agent = _make_codex_agent(tmp_path, monkeypatch) + monkeypatch.setenv("HERMES_CODEX_HARD_TIMEOUT_SECONDS", "0") + + closes: list = [] + dummy_client = SimpleNamespace() + monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client) + monkeypatch.setattr( + agent, "_abort_request_openai_client", + lambda c, reason=None: closes.append(reason), + ) + monkeypatch.setattr( + agent, "_close_request_openai_client", + lambda c, reason=None: closes.append(reason), + ) + + sentinel = SimpleNamespace(ok=True) + + def fake_stream(api_kwargs, client=None, on_first_delta=None): + # No event, but only briefly — well under the (here 60s) stale timeout. + time.sleep(2.0) + return sentinel + + monkeypatch.setattr(agent, "_run_codex_stream", fake_stream) + + large_input = "x" * 44_000 + resp = h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": large_input}) + assert resp is sentinel + assert "codex_ttfb_kill" not in closes + assert "stale_call_kill" not in closes + + +def test_large_codex_request_hard_ceiling_caps_raised_stale_floor(tmp_path, monkeypatch): + """The hard ceiling must cap the raised stale floor (openai-codex can push + the stale timeout to 1200s at >100k tokens). A large silent stall must die + at the ceiling, proving the min() wins over the floor. + """ + from agent import chat_completion_helpers as h + + agent = _make_codex_agent(tmp_path, monkeypatch) + monkeypatch.setenv("HERMES_CODEX_HARD_TIMEOUT_SECONDS", "4") + # Force the >100k-token tier so openai_codex_stale_timeout_floor returns 1200s. + monkeypatch.setattr( + agent, "_compute_non_stream_stale_timeout", lambda *a, **k: 1200.0 + ) + + closes: list = [] + dummy_client = SimpleNamespace() + monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client) + monkeypatch.setattr( + agent, "_abort_request_openai_client", + lambda c, reason=None: closes.append(reason), + ) + monkeypatch.setattr( + agent, "_close_request_openai_client", + lambda c, reason=None: closes.append(reason), + ) + + stop = {"flag": False} + + def fake_hang(api_kwargs, client=None, on_first_delta=None): + deadline = time.time() + 200 + while time.time() < deadline and not stop["flag"] and not agent._interrupt_requested: + time.sleep(0.02) + raise RuntimeError("connection closed") + + monkeypatch.setattr(agent, "_run_codex_stream", fake_hang) + + huge_input = "x" * 500_000 # ~125k tokens → stale floor 1200s + t0 = time.time() + try: + with pytest.raises(TimeoutError): + h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": huge_input}) + elapsed = time.time() - t0 + assert elapsed < 40, f"hard ceiling lost to stale floor: {elapsed:.1f}s" + assert "stale_call_kill" in closes, f"stale kill expected, got {closes}" + finally: + stop["flag"] = True