diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 12c80a8bfb71..dd05ce1e4795 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -857,10 +857,19 @@ def run_conversation( if moa_config: try: + from agent.message_content import flatten_message_text as _flatten_mt from agent.moa_loop import _preset_temperature, aggregate_moa_context _moa_context = aggregate_moa_context( - user_prompt=original_user_message if isinstance(original_user_message, str) else str(original_user_message), + user_prompt=( + original_user_message + if isinstance(original_user_message, str) + # Multimodal / decorated content list: extract the + # visible text instead of str()-ing a Python repr of + # the parts (which would leak base64 image payloads + # into the aggregator prompt). + else _flatten_mt(original_user_message) + ), api_messages=api_messages, reference_models=moa_config.get("reference_models") or [], aggregator=moa_config.get("aggregator") or {}, @@ -874,6 +883,14 @@ def run_conversation( _base = _msg.get("content", "") if isinstance(_base, str): _msg["content"] = _base + "\n\n" + _moa_context + elif isinstance(_base, list): + # Multimodal user turn (text + image parts): + # append the MoA context as a trailing text + # part instead of silently dropping it. + _msg["content"] = [ + *_base, + {"type": "text", "text": "\n\n" + _moa_context}, + ] break except Exception as _moa_exc: logger.warning("MoA context aggregation failed: %s", _moa_exc) diff --git a/agent/moa_loop.py b/agent/moa_loop.py index cd7fa3ea4ea2..267f33dd7e5a 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -14,6 +14,7 @@ from concurrent.futures import ThreadPoolExecutor from typing import Any from agent.auxiliary_client import call_llm +from agent.message_content import flatten_message_text from agent.transports import get_transport logger = logging.getLogger(__name__) @@ -470,11 +471,36 @@ def _reference_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]: for msg in messages: role = msg.get("role") content = msg.get("content") - text = content if isinstance(content, str) else "" + # Flatten structured content (lists of parts) to visible text. Content + # arrives as a list — not a string — in two common cases: + # 1. Anthropic prompt-cache decoration: conversation_loop runs + # apply_anthropic_cache_control BEFORE the MoA facade, converting + # string content to [{"type": "text", "text": ..., "cache_control": + # ...}]. A str-only read here flattened the user's ENTIRE prompt to + # "" — Claude references then 400'd ("messages: at least one + # message is required") while tolerant models answered "no user + # request is present". + # 2. Multimodal turns (pasted image → text + image_url parts) and + # multimodal tool results (screenshots). + # flatten_message_text extracts the text parts and skips image parts, + # and returns strings unchanged — so a decorated and an undecorated + # transcript produce a byte-identical advisory view (which keeps the + # advisory prefix stable across iterations for advisor prompt caching). + text = flatten_message_text(content) if role == "system": continue if role == "user": + if not text.strip() and content not in (None, "", []): + # Structured content with no extractable text (e.g. an + # image-only turn). Emitting an empty user message would be + # dropped/rejected by strict providers (Anthropic 400s on + # empty text blocks — the original "closed" preset failure + # mode), and silently skipping the turn would break + # user/assistant alternation in the advisory view. Substitute + # a placeholder so the reference knows a non-text turn + # happened. + text = "[user sent non-text content (e.g. an image attachment)]" if text.strip(): last_user_content = text rendered.append({"role": "user", "content": text}) @@ -517,8 +543,10 @@ def _reference_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]: if last_user_content is not None: return [{"role": "user", "content": last_user_content}] for msg in reversed(messages): - if msg.get("role") == "user" and isinstance(msg.get("content"), str): - return [{"role": "user", "content": msg["content"]}] + if msg.get("role") == "user": + fallback_text = flatten_message_text(msg.get("content")) + if fallback_text.strip(): + return [{"role": "user", "content": fallback_text}] return rendered @@ -671,13 +699,28 @@ def _attach_reference_guidance(agg_messages: list[dict[str, Any]], guidance: str Appending at the very end keeps the ``[system][task][tool-history]`` prefix stable and cache-reusable (only the new block re-prefills), and gives the aggregator the references with recency. Merge into the last message only when - it is already a trailing string ``user`` turn (plain chat — still at the end). + it is already a trailing ``user`` turn (plain chat — still at the end). + + A trailing user turn's content may be a STRING or a LIST of content parts — + Anthropic prompt-cache decoration (which runs before the MoA facade) + converts string content to ``[{"type": "text", ..., "cache_control": ...}]``, + and multimodal turns are lists natively. Both shapes are merged in place: + appending a new text part AFTER the cache_control-marked part keeps the + cached prefix byte-stable (the marker still terminates it) while the + turn-varying guidance rides outside the cached span. Appending a SEPARATE + user message here instead would produce two consecutive user turns — + strict providers reject that. """ last = agg_messages[-1] if agg_messages else None - if last is not None and last.get("role") == "user" and isinstance(last.get("content"), str): - last["content"] = last["content"] + "\n\n" + guidance - else: - agg_messages.append({"role": "user", "content": guidance}) + if last is not None and last.get("role") == "user": + last_content = last.get("content") + if isinstance(last_content, str): + last["content"] = last_content + "\n\n" + guidance + return + if isinstance(last_content, list): + last["content"] = [*last_content, {"type": "text", "text": "\n\n" + guidance}] + return + agg_messages.append({"role": "user", "content": guidance}) class MoAChatCompletions: diff --git a/tests/run_agent/test_moa_loop_mode.py b/tests/run_agent/test_moa_loop_mode.py index 04c7d51ceb10..094278541762 100644 --- a/tests/run_agent/test_moa_loop_mode.py +++ b/tests/run_agent/test_moa_loop_mode.py @@ -1059,3 +1059,147 @@ def test_reference_guidance_merges_into_trailing_user_in_plain_chat(): assert len(messages) == 2 assert messages[-1]["role"] == "user" assert messages[-1]["content"] == "hello\n\nREFERENCE BLOCK" + + +def test_reference_messages_flattens_cache_decorated_content(): + """Cache-decorated turns (content-part lists) must not blind the references. + + conversation_loop runs apply_anthropic_cache_control BEFORE the MoA facade + when the preset's aggregator is a cache-honoring Claude route (post-#57675). + That converts string content into [{"type": "text", "text": ..., + "cache_control": ...}] lists. The advisory view previously read only string + content, so the user's ENTIRE prompt flattened to "" — Claude references + then 400'd ("messages: at least one message is required") while tolerant + models answered "no user request is present" (live incident, Jul 14 2026, + preset "closed", session 20260714_001520_28157b). + """ + from agent.moa_loop import _reference_messages + from agent.prompt_caching import apply_anthropic_cache_control + + plain = [ + {"role": "system", "content": "hermes system prompt"}, + {"role": "user", "content": "Can we get codex usage resets into hermes?"}, + ] + decorated = apply_anthropic_cache_control(plain, native_anthropic=False) + # Premise: decoration really converts the user turn to a content-part list. + assert isinstance(decorated[1]["content"], list) + + view = _reference_messages(decorated) + + assert view == [ + {"role": "user", "content": "Can we get codex usage resets into hermes?"} + ] + # Invariant: decorated and undecorated transcripts produce the SAME + # advisory view — so decoration can never change what references see, + # and the advisory prefix stays byte-stable for advisor prompt caching. + assert view == _reference_messages(plain) + + +def test_reference_messages_flattens_multimodal_user_turn(): + """Multimodal user turns (text + image parts) keep their text in the view. + + Image parts carry no advisory text and are skipped; the text part must + survive. Previously the whole turn flattened to "". + """ + from agent.moa_loop import _reference_messages + + messages = [ + {"role": "user", "content": [ + {"type": "text", "text": "what is in this screenshot?"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}}, + ]}, + ] + + view = _reference_messages(messages) + + assert view == [{"role": "user", "content": "what is in this screenshot?"}] + # No base64 payload leaks into the advisory view. + assert all("base64" not in m["content"] for m in view) + + +def test_reference_messages_image_only_user_turn_gets_placeholder(): + """An image-only user turn must not become an empty user message. + + Anthropic rejects empty text blocks (the original 400 class) and silently + skipping the turn would misalign user/assistant alternation in the view — + so a placeholder stands in for the non-text content. + """ + from agent.moa_loop import _reference_messages + + messages = [ + {"role": "user", "content": [ + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}}, + ]}, + {"role": "assistant", "content": "I see a diagram."}, + {"role": "user", "content": "now explain it"}, + ] + + view = _reference_messages(messages) + + assert view[0]["role"] == "user" + assert view[0]["content"].strip(), "image-only turn must not be empty" + assert "non-text" in view[0]["content"] + assert view[-1] == {"role": "user", "content": "now explain it"} + + +def test_reference_messages_flattens_structured_assistant_and_tool_content(): + """Assistant and tool turns with content-part lists are flattened too. + + Multimodal tool results (e.g. computer_use screenshots) and adapter-shaped + assistant turns arrive as lists; their text must reach the references and + their image parts must not leak. + """ + from agent.moa_loop import _reference_messages + + messages = [ + {"role": "user", "content": "check the screen"}, + { + "role": "assistant", + "content": [{"type": "text", "text": "taking a screenshot"}], + "tool_calls": [{"id": "c1", "function": {"name": "capture", "arguments": "{}"}}], + }, + {"role": "tool", "tool_call_id": "c1", "content": [ + {"type": "text", "text": "screenshot captured: login page visible"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,BBBB"}}, + ]}, + ] + + view = _reference_messages(messages) + + joined = "\n".join(m["content"] for m in view) + assert "taking a screenshot" in joined + assert "[called tool: capture(" in joined + assert "[tool result: screenshot captured: login page visible]" in joined + assert "BBBB" not in joined + assert view[-1]["role"] == "user" + + +def test_reference_guidance_appends_text_part_to_decorated_trailing_user(): + """A cache-decorated trailing user turn still receives the guidance block. + + Decoration converts the trailing user turn to a content-part list; the + guidance must be appended as a NEW text part AFTER the cache_control-marked + part (cached prefix stays byte-stable, no consecutive-user-turn 400s), not + silently dropped and not added as a second user message. + """ + from agent.moa_loop import _attach_reference_guidance + + marked_part = { + "type": "text", + "text": "hello", + "cache_control": {"type": "ephemeral"}, + } + messages = [ + {"role": "system", "content": "system prompt"}, + {"role": "user", "content": [dict(marked_part)]}, + ] + _attach_reference_guidance(messages, "REFERENCE BLOCK") + + # No extra message (would break user/user alternation). + assert len(messages) == 2 + content = messages[-1]["content"] + assert isinstance(content, list) and len(content) == 2 + # The cache-marked part is byte-identical (prefix stability). + assert content[0] == marked_part + # The guidance rides as a trailing text part outside the cached span. + assert content[1] == {"type": "text", "text": "\n\nREFERENCE BLOCK"}