From 28b6bd657020334ef8e84eccd24f3f2aae98c46c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 22 Jul 2026 04:36:32 -0700 Subject: [PATCH] fix(api): preserve compaction summaries at any history position during Responses auto-truncation The gateway /compress path can force a user-leading layout that leaves the compaction summary after a retained system head, so scanning only the leading block misses it. Preserve marker-carrying messages wherever they sit, filling the remaining budget with the most recent other messages, and cover the non-leading position with a test. --- gateway/platforms/api_server.py | 41 +++++++++++++++++----------- tests/gateway/test_api_server.py | 47 ++++++++++++++++++++++++++++++++ 2 files changed, 72 insertions(+), 16 deletions(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index ddbcdf1bc6b7..43c3486806f9 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -200,28 +200,37 @@ def _auto_truncate_response_history( *, limit: int = RESPONSES_AUTO_TRUNCATION_HISTORY_LIMIT, ) -> List[Dict[str, Any]]: - """Keep recent Responses history without dropping the compaction handoff.""" + """Keep recent Responses history without dropping the compaction handoff. + + Compaction summaries are preserved wherever they sit in the history — + the gateway /compress path can leave them after a retained system head + (see ``context_compressor`` force-user-leading handling), so a + leading-block-only scan would silently drop them. + """ if limit <= 0 or len(conversation_history) <= limit: return conversation_history - leading_summaries: List[Dict[str, Any]] = [] - first_conversation_index = 0 - for message in conversation_history: - if not _is_compressed_summary_message(message): - break - leading_summaries.append(message) - first_conversation_index += 1 - - if not leading_summaries: + summary_indices = [ + index + for index, message in enumerate(conversation_history) + if _is_compressed_summary_message(message) + ] + if not summary_indices: return conversation_history[-limit:] - preserved = leading_summaries[:limit] - remaining = limit - len(preserved) - if remaining <= 0: - return preserved + kept_indices = set(summary_indices[:limit]) + remaining = limit - len(kept_indices) + if remaining > 0: + summary_index_set = set(summary_indices) + for index in range(len(conversation_history) - 1, -1, -1): + if index in summary_index_set: + continue + kept_indices.add(index) + remaining -= 1 + if remaining <= 0: + break - recent_tail = conversation_history[first_conversation_index:][-remaining:] - return preserved + recent_tail + return [conversation_history[index] for index in sorted(kept_indices)] def _normalize_chat_content( diff --git a/tests/gateway/test_api_server.py b/tests/gateway/test_api_server.py index 0010b5a841f3..53beee7fcd3e 100644 --- a/tests/gateway/test_api_server.py +++ b/tests/gateway/test_api_server.py @@ -3580,6 +3580,53 @@ class TestTruncation: assert history[1]["content"] == "msg 50" assert history[-1]["content"] == "msg 148" + @pytest.mark.asyncio + async def test_truncation_auto_preserves_non_leading_compaction_summary(self, adapter): + """A summary sitting after a retained system head must survive too. + + The gateway /compress path can force a user-leading layout that + leaves the compaction summary after a kept system message, so the + preservation predicate must not assume the summary is at index 0. + """ + mock_result = {"final_response": "OK", "messages": [], "api_calls": 1} + + system_head = {"role": "system", "content": "You are a helpful agent."} + summary = { + "role": "user", + "content": "[CONTEXT COMPACTION — REFERENCE ONLY]\nEarlier work.", + "_compressed_summary": True, + } + long_history = [system_head, summary] + [ + {"role": "user", "content": f"msg {i}"} + for i in range(148) + ] + adapter._response_store.put("resp_summary_mid", { + "response": {"id": "resp_summary_mid", "object": "response"}, + "conversation_history": long_history, + "instructions": None, + }) + + app = _create_app(adapter) + async with TestClient(TestServer(app)) as cli: + with patch.object(adapter, "_run_agent", new_callable=AsyncMock) as mock_run: + mock_run.return_value = (mock_result, {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0}) + resp = await cli.post( + "/v1/responses", + json={ + "model": "hermes-agent", + "input": "follow up", + "previous_response_id": "resp_summary_mid", + "truncation": "auto", + }, + ) + + assert resp.status == 200 + history = mock_run.call_args.kwargs["conversation_history"] + assert len(history) == 100 + assert history[0] == summary + assert history[1]["content"] == "msg 49" + assert history[-1]["content"] == "msg 147" + @pytest.mark.asyncio async def test_no_truncation_keeps_full_history(self, adapter): """Without truncation=auto, long history is passed as-is."""