From aef0fe6f27a538fd44b57e9a000f548f9e1f80a8 Mon Sep 17 00:00:00 2001 From: nanami7777777 <178757933+nanami7777777@users.noreply.github.com> Date: Fri, 17 Jul 2026 21:45:00 +0800 Subject: [PATCH] fix: handle multimodal content in interim assistant text and avoid retrying local processing errors (#66267) --- agent/agent_runtime_helpers.py | 21 +++++++++++-- agent/conversation_loop.py | 57 +++++++++++++++++++++++++++++++--- run_agent.py | 6 +++- 3 files changed, 75 insertions(+), 9 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index d1803048babf..db0c48275cc8 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -643,16 +643,16 @@ def strip_think_blocks(agent, content: str) -> str: """Remove reasoning/thinking blocks from content, returning only visible text. Handles four cases: - 1. Closed tag pairs (````) — the common path when + 1. Closed tag pairs (`` 执教… ``) — the common path when the provider emits complete reasoning blocks. 2. Unterminated open tag at a block boundary (start of text or after a newline) — e.g. MiniMax M2.7 / NIM endpoints where the closing tag is dropped. Everything from the open tag to end of string is stripped. The block-boundary check mirrors ``gateway/stream_consumer.py``'s filter so models that mention - ```` in prose aren't over-stripped. + `` 执教`` in prose aren't over-stripped. 3. Stray orphan open/close tags that slip through. - 4. Tag variants: ````, ````, ````, + 4. Tag variants: `` 执教``, ````, ````, ````, ```` (Gemma 4), all case-insensitive. @@ -670,6 +670,21 @@ def strip_think_blocks(agent, content: str) -> str: after punctuation and carries a ``name="..."`` attribute) so prose mentions like "Use in JavaScript" are preserved. """ + # Defensive coercion: callers may pass multimodal content-parts lists + # (e.g. after vision turns or context compaction). Extract any text + # blocks and drop non-text parts before running regexes. + if isinstance(content, list): + parts = [] + for part in content: + if isinstance(part, str): + parts.append(part) + elif isinstance(part, dict) and part.get("type") == "text": + text = part.get("text") + if isinstance(text, str): + parts.append(text) + content = "\n".join(parts) + elif not isinstance(content, str): + content = str(content) if content is not None else "" if not content: return "" # Coerce non-string content to text before any regex runs. Providers diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 21d065fcbf78..ffe114e15842 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -24,6 +24,7 @@ import re import ssl import threading import time +import traceback import uuid from typing import Any, Dict, List, Optional @@ -5597,7 +5598,44 @@ def run_conversation( break except Exception as e: - error_msg = f"Error during OpenAI-compatible API call #{api_call_count}: {str(e)}" + # Phase-aware error classification. The huge outer try/except spans + # both the actual API request and all local post-processing of the + # returned assistant message. Deterministic local bugs (e.g. + # passing a multimodal content list into a regex helper after a + # vision turn or context compaction) should not be retried: they + # will fail identically on every iteration and only burn the + # iteration budget. We classify an error as local by inspecting the + # traceback: if the exception propagated through any of the known + # local post-processing helpers and never entered the interruptible + # API-call helpers, it is almost certainly a local processing bug. + # (#66267) + _local_processing_modules = { + "agent_runtime_helpers", + "conversation_loop", + "message_content", + "run_agent", + } + _api_call_modules = { + "chat_completion_helpers", + } + + tb_stack = traceback.extract_tb(e.__traceback__) + tb_module_names = { + os.path.splitext(os.path.basename(frame.filename))[0] + for frame in tb_stack + } + _hit_local = bool(tb_module_names & _local_processing_modules) + _hit_api = bool(tb_module_names & _api_call_modules) + + _is_local_processing_error = _hit_local and not _hit_api + + if _is_local_processing_error: + error_msg = ( + f"Error during local message processing after " + f"OpenAI-compatible API call #{api_call_count}: {str(e)}" + ) + else: + error_msg = f"Error during OpenAI-compatible API call #{api_call_count}: {str(e)}" try: print(f"❌ {error_msg}") except (OSError, ValueError): @@ -5644,10 +5682,19 @@ def run_conversation( # message pollutes history, burns tokens, and risks violating # role-alternation invariants. - # If we're near the limit, break to avoid infinite loops - if api_call_count >= agent.max_iterations - 1: - _turn_exit_reason = f"error_near_max_iterations({error_msg[:80]})" - final_response = f"I apologize, but I encountered repeated errors: {error_msg}" + # If we're near the limit, break to avoid infinite loops. + # Local processing errors are deterministic — stop immediately + # rather than retrying until the budget is exhausted. + if ( + _is_local_processing_error + or api_call_count >= agent.max_iterations - 1 + ): + if _is_local_processing_error: + _turn_exit_reason = f"local_processing_error({error_msg[:80]})" + final_response = f"I apologize, but I encountered an error while processing the model response: {error_msg}" + else: + _turn_exit_reason = f"error_near_max_iterations({error_msg[:80]})" + final_response = f"I apologize, but I encountered repeated errors: {error_msg}" # Append as assistant so the history stays valid for # session resume (avoids consecutive user messages). messages.append({"role": "assistant", "content": final_response}) diff --git a/run_agent.py b/run_agent.py index e48e83f8f7dc..8a92981e1dd2 100644 --- a/run_agent.py +++ b/run_agent.py @@ -148,6 +148,7 @@ from tools.browser_tool import cleanup_browser from agent.memory_manager import sanitize_context from agent.error_classifier import FailoverReason from agent.redact import redact_sensitive_text +from agent.message_content import flatten_message_text from agent.model_metadata import ( estimate_request_tokens_rough, # noqa: F401 # re-exported for tests that mock.patch("run_agent.estimate_request_tokens_rough") is_local_endpoint, @@ -4823,12 +4824,15 @@ class AIAgent: response can contain both commentary and a partial/final-answer message while tools are still pending; treating top-level content as progress in that shape leaks the answer before the tool call runs. + + Content may be a string or a structured parts list (e.g. after vision + turns or context compaction), so flatten it before stripping reasoning. """ visible = self._extract_codex_interim_visible_text(assistant_msg) if visible: return visible content = assistant_msg.get("content") - return self._strip_think_blocks(content or "").strip() + return self._strip_think_blocks(flatten_message_text(content)).strip() def _interim_text_was_delivered(self, text: str) -> bool: normalized = self._normalize_interim_visible_text(text)