From cf0c42fa0b27bd453bcab5d8257544ae64b5c963 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Mon, 27 Jul 2026 14:57:11 -0700 Subject: [PATCH] fix(agent): never replay chain-of-thought in the active-turn redirect checkpoint MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A /steer redirect during a thinking phase serialized the streamed reasoning into the persisted assistant checkpoint ('Reasoning shown before the interruption: ...'). An assistant turn exposing its own chain-of-thought reads to Anthropic's output classifier as reasoning-injection/prefill jailbreak, so every subsequent call on the session deterministically returned 'Provider returned an empty response' — and because the checkpoint is persisted and replayed, no retry, nudge, or empty-recovery branch could ever escape it. Four sessions were permanently bricked this way in the week of Jul 21-27 (42+ blocked calls; every reasoning-free checkpoint that week was untouched — same mechanism as the prefill.json incident). Class fix: streamed reasoning is now display-only state. The _current_streamed_reasoning_text accumulator is removed entirely (producer in _fire_reasoning_delta, resets, and init), so no future path can serialize CoT into replayable content. The checkpoint keeps only the visible response text; the model regenerates its reasoning on the retried turn. Invariant documented in _apply_active_turn_redirect. Regression tests: CoT never appears in either checkpoint shape, reasoning-only interrupts produce a bare checkpoint, reasoning deltas stay display-only. --- agent/agent_init.py | 6 ---- agent/conversation_loop.py | 28 +++++++++------ run_agent.py | 10 ------ tests/run_agent/test_run_agent.py | 14 ++++++-- tests/run_agent/test_steer.py | 58 ++++++++++++++++++++++++++----- 5 files changed, 79 insertions(+), 37 deletions(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index 5c022f4f6fb7..22f617b7f1c6 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -958,12 +958,6 @@ def init_agent( agent._stream_writer_tls = threading.local() agent._stream_writer_dropped = 0 - # Displayed reasoning text streamed during the current model response, - # captured only when a surface consumed it via a reasoning callback. Used - # by active-turn redirect to checkpoint what the user actually saw without - # ever persisting hidden provider reasoning. - agent._current_streamed_reasoning_text = "" - # Optional current-turn user-message override used when the API-facing # user message intentionally differs from the persisted transcript # (e.g. CLI voice mode adds a temporary prefix for the live call only). diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 6b5d4d6ad975..2efad4e95801 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -121,22 +121,31 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text Incomplete provider reasoning blocks are not valid replay items (Anthropic signs them; Responses reasoning items require their following output). - Preserve only what Hermes actually displayed, demoted to ordinary text, - then add the correction as a real user message. This keeps role alternation + Preserve only the *visible* response text, demoted to ordinary text, then + add the correction as a real user message. This keeps role alternation valid and leaves every previously cached message byte-for-byte unchanged. + + INVARIANT — raw chain-of-thought must never be serialized into replayable + message content. Streamed reasoning is display-only state: it may be shown + live, but it does not re-enter the transcript as assistant (or user) text. + An assistant turn whose content inlines its own chain-of-thought reads to + Anthropic's output classifier as reasoning-injection/prefill jailbreak, + and because the poisoned checkpoint is persisted and replayed on every + subsequent call, the session dies permanently with deterministic + "Provider returned an empty response" storms that no retry, nudge, or + empty-recovery branch can escape (July 2026: four sessions bricked this + way; every reasoning-free checkpoint that week was untouched — same + mechanism as the ~/.hermes/prefill.json incident, 20/20 blocked with + assistant-exposed CoT vs 0/20 without). The interrupted reasoning was + incomplete by definition; the model regenerates it on the retried turn. + If a future path needs to preserve interrupted thinking, carry it in a + provider-gated reasoning *field*, never in content. """ - reasoning = str( - getattr(agent, "_current_streamed_reasoning_text", "") or "" - ).strip() visible = agent._strip_think_blocks( getattr(agent, "_current_streamed_assistant_text", "") or "" ).strip() checkpoint_parts = ["[This response was interrupted by a user correction.]"] - if reasoning: - checkpoint_parts.extend( - ["Reasoning shown before the interruption:", reasoning] - ) if visible: checkpoint_parts.extend( ["Visible response before the interruption:", visible] @@ -159,7 +168,6 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text messages.append({"role": "user", "content": text}) agent._current_streamed_assistant_text = "" - agent._current_streamed_reasoning_text = "" agent._stream_needs_break = True diff --git a/run_agent.py b/run_agent.py index fe8db996955d..71176691839e 100644 --- a/run_agent.py +++ b/run_agent.py @@ -5324,7 +5324,6 @@ class AIAgent: pass self._record_streamed_assistant_text(tail) self._current_streamed_assistant_text = "" - self._current_streamed_reasoning_text = "" def _record_streamed_assistant_text(self, text: str) -> None: """Accumulate visible assistant text emitted through stream callbacks.""" @@ -5665,15 +5664,6 @@ class AIAgent: cb(text) except Exception: pass - else: - # Only checkpoint reasoning that a surface actually displayed. - # show_reasoning=false leaves the callback unset, so hidden - # provider thinking never becomes visible transcript content. - if isinstance(text, str) and text: - self._current_streamed_reasoning_text = ( - getattr(self, "_current_streamed_reasoning_text", "") - + text - ) def _fire_tool_gen_started(self, tool_name: str) -> None: """Notify display layer that the model is generating tool call arguments. diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index f44dca8882e0..a9cac898e31c 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -4894,7 +4894,8 @@ class TestRunConversation: } def test_redirect_during_thinking_retries_same_turn_with_context(self, agent): - """A corrective follow-up keeps displayed reasoning and does not end the turn.""" + """A corrective follow-up does not end the turn, and displayed reasoning + never re-enters the transcript (classifier-poisoning guard).""" self._setup_agent(agent) agent.reasoning_callback = lambda _text: None final = _mock_response(content="Using Postgres instead.", finish_reason="stop") @@ -4936,7 +4937,11 @@ class TestRunConversation: ] checkpoint = replay[-2]["content"] assert "interrupted by a user correction" in checkpoint - assert "I should implement this with SQLite." in checkpoint + # Displayed chain-of-thought must NOT be replayed: an assistant turn + # inlining its own reasoning trips Anthropic's output classifier and + # bricks the session with deterministic empty responses (July 2026). + assert "I should implement this with SQLite." not in checkpoint + assert "Reasoning shown before the interruption" not in checkpoint assert replay[-1]["content"] == "No, use Postgres instead." assert agent._pending_redirect is None assert any( @@ -5018,7 +5023,10 @@ class TestRunConversation: assert results["result"]["completed"] is True assert results["result"]["final_response"] == "Corrected answer." checkpoint = results["result"]["messages"][-3] - assert "Following the original approach." in checkpoint["content"] + assert "interrupted by a user correction" in checkpoint["content"] + # Displayed reasoning is display-only — replaying it as assistant + # content trips Anthropic's output classifier (July 2026 brickings). + assert "Following the original approach." not in checkpoint["content"] assert results["result"]["messages"][-2]["content"] == ( "Use the corrected approach." ) diff --git a/tests/run_agent/test_steer.py b/tests/run_agent/test_steer.py index ebf84c617d4b..7fbd40b3be89 100644 --- a/tests/run_agent/test_steer.py +++ b/tests/run_agent/test_steer.py @@ -35,7 +35,6 @@ def _bare_agent() -> AIAgent: agent._active_children_lock = threading.Lock() agent._tool_worker_threads = None agent._tool_worker_threads_lock = None - agent._current_streamed_reasoning_text = "" agent._current_streamed_assistant_text = "" agent._stream_needs_break = False agent._strip_think_blocks = lambda content: content @@ -125,14 +124,20 @@ class TestActiveTurnRedirect: assert agent.redirect("too late") is False assert agent._pending_redirect is None - def test_hidden_reasoning_is_not_checkpointed(self): + def test_reasoning_deltas_are_display_only(self): + """Streamed reasoning must never accumulate into replayable transcript + state — an assistant checkpoint that inlines chain-of-thought trips + Anthropic's output classifier and permanently bricks the session + (deterministic empty-response storms on every replay).""" agent = _bare_agent() - agent.reasoning_callback = None - agent._current_streamed_reasoning_text = "" + seen = [] + agent.reasoning_callback = seen.append - agent._fire_reasoning_delta("private provider thinking") + agent._fire_reasoning_delta("visible provider thinking") - assert agent._current_streamed_reasoning_text == "" + # Displayed to the surface, but never checkpointed anywhere. + assert seen == ["visible provider thinking"] + assert not getattr(agent, "_current_streamed_reasoning_text", "") def test_response_completion_before_redirect_lock_rejects_correction(self): agent = _bare_agent() @@ -227,7 +232,6 @@ class TestActiveTurnRedirectCheckpoint: from agent.conversation_loop import _apply_active_turn_redirect agent = _bare_agent() - agent._current_streamed_reasoning_text = "Shown reasoning." agent._current_streamed_assistant_text = "Visible draft." messages = [ {"role": "user", "content": "start"}, @@ -240,10 +244,48 @@ class TestActiveTurnRedirectCheckpoint: assert messages[-1]["role"] == "user" assert messages[-1]["content"].endswith("Use Postgres instead.") assert sum(1 for m in messages if m["role"] == "assistant") == 1 - assert "Shown reasoning." in messages[-1]["content"] assert "Visible draft." in messages[-1]["content"] assert "Context from the interrupted assistant response" in messages[-1]["content"] + def test_checkpoint_never_replays_chain_of_thought(self): + """Raw CoT serialized into checkpoint content reads to Anthropic's + output classifier as reasoning-injection; because the checkpoint is + persisted and replayed on every later call, one redirect during a + thinking phase permanently bricked sessions with deterministic + empty-response storms (July 2026). Reasoning must never appear in + replayable content — in either the assistant-checkpoint or the + merged-user-correction shape.""" + from agent.conversation_loop import _apply_active_turn_redirect + + for tail_role in ("user", "assistant"): + agent = _bare_agent() + # Simulate a surface having displayed reasoning this turn. + agent._current_streamed_reasoning_text = "SECRET chain of thought." + agent._current_streamed_assistant_text = "Visible draft." + messages = [{"role": "user", "content": "start"}] + if tail_role == "assistant": + messages.append({"role": "assistant", "content": "committed"}) + + _apply_active_turn_redirect(agent, messages, "Change course.") + + serialized = "".join(str(m.get("content", "")) for m in messages) + assert "SECRET chain of thought." not in serialized + assert "Reasoning shown before the interruption" not in serialized + assert "Visible draft." in serialized + + def test_checkpoint_omits_reasoning_label_when_nothing_visible(self): + from agent.conversation_loop import _apply_active_turn_redirect + + agent = _bare_agent() + agent._current_streamed_reasoning_text = "thinking only, no text yet" + messages = [{"role": "user", "content": "start"}] + + _apply_active_turn_redirect(agent, messages, "New direction.") + + checkpoint = messages[-2]["content"] + assert checkpoint == "[This response was interrupted by a user correction.]" + assert messages[-1]["content"] == "New direction." + class TestSteerInjection: def test_appends_to_last_tool_result(self):