fix: bound compression summary input

This commit is contained in:
Cluster2 2026-05-18 00:26:58 -04:00 committed by Teknium
parent fa4800414c
commit 80ece3867b
2 changed files with 64 additions and 1 deletions

View file

@ -2440,6 +2440,7 @@ class ContextCompressor(ContextEngine):
_CONTENT_TAIL = 1500 # chars kept from the end
_TOOL_ARGS_MAX = 1500 # tool call argument chars
_TOOL_ARGS_HEAD = 1200 # kept from the start of tool args
_SUMMARY_INPUT_MAX_CHARS = 160_000 # total serialized turns sent to aux summarizer
def _serialize_for_summary(self, turns: List[Dict[str, Any]]) -> str:
"""Serialize conversation turns into labeled text for the summarizer.
@ -2735,6 +2736,39 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
summary = summary[: _FALLBACK_SUMMARY_MAX_CHARS - 42].rstrip() + "\n...[fallback summary truncated]"
return summary
@classmethod
def _bound_summary_input(cls, content: str) -> str:
"""Cap total summarizer input while preserving beginning and recent tail.
Per-message truncation alone is not enough for very long sessions: a
compression window with hundreds of messages can still produce a huge
single prompt that slow auxiliary backends time out on. Keep both edges
because the beginning often has task setup and the tail has the most
recent state; explicitly mark the omitted middle so the summarizer knows
context was intentionally compressed before it saw the prompt.
"""
if len(content) <= cls._SUMMARY_INPUT_MAX_CHARS:
return content
marker_template = (
"\n\n...[summary input truncated: omitted "
"{omitted:,} chars from the middle to keep compression prompt bounded]...\n\n"
)
# Estimate once, then rebuild with the exact omitted span after the
# head/tail split is known. The second marker can differ by a few chars
# if the comma-formatted number changes width, so recompute once.
marker = marker_template.format(omitted=len(content))
remaining = max(cls._SUMMARY_INPUT_MAX_CHARS - len(marker), 0)
head_chars = int(remaining * 0.45)
tail_chars = remaining - head_chars
omitted = max(len(content) - head_chars - tail_chars, 0)
marker = marker_template.format(omitted=omitted)
remaining = max(cls._SUMMARY_INPUT_MAX_CHARS - len(marker), 0)
head_chars = int(remaining * 0.45)
tail_chars = remaining - head_chars
tail = content[-tail_chars:].lstrip() if tail_chars else ""
return content[:head_chars].rstrip() + marker + tail
def _fallback_to_main_for_compression(self, e: Exception, reason: str) -> None:
"""Switch from a separate ``summary_model`` back to the main model.
@ -2807,7 +2841,9 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
self._previous_summary = _redact_compaction_text(self._previous_summary)
summary_budget = self._compute_summary_budget(turns_to_summarize)
content_to_summarize = self._serialize_for_summary(turns_to_summarize)
content_to_summarize = self._bound_summary_input(
self._serialize_for_summary(turns_to_summarize)
)
_sanitized_memory_context = sanitize_memory_context(memory_context)
_serialized_memory_context = json.dumps(
_sanitized_memory_context,

View file

@ -3973,3 +3973,30 @@ class TestDoubleCompactionSummaryRole:
"summary of earlier turns" in (m.get("content") or "")
for m in result
)
class TestSummaryPromptBounding:
def test_oversized_summary_prompt_is_bounded_and_preserves_edges(self):
mock_response = MagicMock()
mock_response.choices = [MagicMock()]
mock_response.choices[0].message.content = "bounded summary"
with patch("agent.context_compressor.get_model_context_length", return_value=272000):
c = ContextCompressor(model="test", quiet_mode=True)
messages = [
{"role": "user", "content": f"turn-{i}-" + ("x" * 6000)}
for i in range(80)
]
messages[0]["content"] = "FIRST_SENTINEL " + messages[0]["content"]
messages[-1]["content"] = "LAST_SENTINEL " + messages[-1]["content"]
with patch("agent.context_compressor.call_llm", return_value=mock_response) as mock_call:
summary = c._generate_summary(messages)
prompt = mock_call.call_args.kwargs["messages"][0]["content"]
assert summary.startswith(SUMMARY_PREFIX)
assert len(prompt) < 180_000
assert "summary input truncated" in prompt
assert "FIRST_SENTINEL" in prompt
assert "LAST_SENTINEL" in prompt