fix(agent): decay protected summaries after restart

protect_first_n decay state (compression_count / _previous_summary) is
in-memory only, so a gateway restart re-protected the persisted handoff
summary and head fossils, growing the head unboundedly across
restart+compaction cycles (#57814).

_effective_protect_first_n now probes a bounded resumed-head window for
a persisted handoff summary (by metadata or content prefix) and decays
protection when one is found, before compress_start is computed. The
first post-restart compaction self-heals: stacked summary fossils are
folded into the next summary prompt instead of preserved verbatim,
rehydration is rolled back on abort, deterministic fallback carries a
bounded redacted previous-summary snapshot, and forced user-leading
merged summaries keep the live tail request after the summary end
marker so they stay rehydratable.

Squashed reapply of the 12-commit series from PR #57835 onto current
main (branch was 1210 commits behind; single add/add conflict with the
task-snapshot grounding helpers resolved by keeping both).

Fixes #57814.
This commit is contained in:
harjoth 2026-07-22 05:43:51 -07:00 committed by Teknium
parent cbc1054e23
commit 08ea88f4f7
3 changed files with 617 additions and 54 deletions

View file

@ -288,6 +288,12 @@ _HISTORICAL_SUMMARY_PREFIXES = (
"config, etc.) may reflect work described here — avoid repeating it:",
)
# Restart handoff detection should be early and bounded: it needs to catch the
# restored protected head plus a small cluster of already-stacked handoff/ack
# turns, but it must not treat arbitrary summary-looking live-tail messages as
# proof that this is a resumed compacted session.
_RESTART_HANDOFF_PROBE_EXTRA_MESSAGES = 4
# Minimum tokens for the summary output
_MIN_SUMMARY_TOKENS = 2000
# Proportion of compressed content to allocate for summary
@ -318,6 +324,7 @@ _SUMMARY_FAILURE_COOLDOWN_SECONDS = 600
# only meant to preserve continuity anchors from the dropped window, not to
# become another unbounded transcript copy after the LLM summarizer failed.
_FALLBACK_SUMMARY_MAX_CHARS = 8_000
_FALLBACK_PREVIOUS_SUMMARY_MAX_CHARS = 3_000
_FALLBACK_TURN_MAX_CHARS = 700
_AUTO_FOCUS_MAX_TURNS = 3
_AUTO_FOCUS_TURN_MAX_CHARS = 260
@ -2305,9 +2312,17 @@ class ContextCompressor(ContextEngine):
)
previous_summary_note = ""
if self._previous_summary:
previous_summary = redact_sensitive_text(self._previous_summary.strip())
if len(previous_summary) > _FALLBACK_PREVIOUS_SUMMARY_MAX_CHARS:
previous_summary = (
previous_summary[: _FALLBACK_PREVIOUS_SUMMARY_MAX_CHARS - 45].rstrip()
+ "\n...[previous summary snapshot truncated]"
)
previous_summary_note = (
"\n\nPrevious compaction summary was present and should still be treated as "
"background continuity context, but the latest LLM summary update failed."
"\n\n## Previous Summary Snapshot\n"
f"{previous_summary}\n\n"
"The previous compaction summary above remains background "
"continuity context because the latest LLM summary update failed."
)
reason_text = f" Summary failure reason: {reason}." if reason else ""
@ -2978,11 +2993,15 @@ This compaction should PRIORITISE preserving all information related to the focu
if text.startswith(prefix):
text = text[len(prefix):].lstrip()
break
# Strip the trailing end marker too — a rehydrated handoff body that
# keeps it would leak the boundary directive into the iterative-update
# Strip the end marker too — a rehydrated handoff body that keeps it
# would leak the boundary directive into the iterative-update
# summarizer prompt (and the marker is re-appended on insertion anyway).
if text.endswith(_SUMMARY_END_MARKER):
text = text[: -len(_SUMMARY_END_MARKER)].rstrip()
# Forced user-leading merged summaries keep the live tail request after
# this marker, so truncate at the marker even when it is not the final
# content.
marker_idx = text.find(_SUMMARY_END_MARKER)
if marker_idx >= 0:
text = text[:marker_idx].rstrip()
return text
@classmethod
@ -3081,6 +3100,15 @@ This compaction should PRIORITISE preserving all information related to the focu
"session with no user-authored turns"
)
@classmethod
def _is_context_summary_message(cls, message: Any) -> bool:
"""Return True for summary handoff messages by metadata or content."""
if not isinstance(message, dict):
return False
return cls._has_compressed_summary_metadata(
message
) or cls._is_context_summary_content(message.get("content"))
@classmethod
def _is_blank_user_turn(cls, message: Any) -> bool:
"""Return whether *message* is an empty, non-summary user-role echo."""
@ -3239,6 +3267,24 @@ This compaction should PRIORITISE preserving all information related to the focu
return grounded.strip()
return f"{replacement}{body}".strip()
@classmethod
def _find_context_summaries(
cls,
messages: List[Dict[str, Any]],
start: int,
end: int,
) -> list[tuple[int, str]]:
"""Find handoff summaries inside a compression window."""
summaries: list[tuple[int, str]] = []
for idx in range(start, end):
content = messages[idx].get("content")
if cls._is_context_summary_message(messages[idx]):
summaries.append((
idx,
cls._strip_summary_prefix(_content_text_for_contains(content)),
))
return summaries
@classmethod
def _find_latest_context_summary(
cls,
@ -3247,10 +3293,9 @@ This compaction should PRIORITISE preserving all information related to the focu
end: int,
) -> tuple[Optional[int], str]:
"""Find the newest handoff summary inside a compression window."""
for idx in range(end - 1, start - 1, -1):
content = messages[idx].get("content")
if cls._is_context_summary_content(content):
return idx, cls._strip_summary_prefix(_content_text_for_contains(content))
summaries = cls._find_context_summaries(messages, start, end)
if summaries:
return summaries[-1]
return None, ""
@classmethod
@ -3471,7 +3516,25 @@ This compaction should PRIORITISE preserving all information related to the focu
idx += 1
return idx
def _effective_protect_first_n(self) -> int:
def _restart_handoff_probe_bounds(
self,
messages: List[Dict[str, Any]],
) -> tuple[int, int]:
"""Return the bounded transcript region that can indicate restart decay."""
if not messages or self.protect_first_n <= 0:
return 0, 0
first_non_system = 1 if messages[0].get("role") == "system" else 0
return first_non_system, min(
len(messages),
first_non_system
+ self.protect_first_n
+ _RESTART_HANDOFF_PROBE_EXTRA_MESSAGES,
)
def _effective_protect_first_n(
self,
messages: Optional[List[Dict[str, Any]]] = None,
) -> int:
"""``protect_first_n`` decayed across compression cycles.
``protect_first_n`` keeps the first N non-system messages verbatim so
@ -3483,9 +3546,24 @@ This compaction should PRIORITISE preserving all information related to the focu
once, the early turns are already captured in the handoff summary, so
there's no need to keep re-protecting them: decay to 0 (the system
prompt is still always protected separately by _protect_head_size).
After a restart, infer that decayed state from handoff summaries in the
resumed-head region; disk-persisted restarts rely on the content prefix,
while the metadata branch covers in-process handoff messages.
"""
if self.compression_count >= 1 or self._previous_summary:
return 0
if messages and self.protect_first_n > 0:
# Probe only the early handoff shape created by a resumed compacted
# session. Summary-looking tail content should keep normal tail
# semantics and not decay the initial first-compaction protection.
first_non_system, restart_probe_end = self._restart_handoff_probe_bounds(
messages
)
if any(
self._is_context_summary_message(msg)
for msg in messages[first_non_system:restart_probe_end]
):
return 0
return self.protect_first_n
def _protect_head_size(self, messages: List[Dict[str, Any]]) -> int:
@ -3511,7 +3589,7 @@ This compaction should PRIORITISE preserving all information related to the focu
head = 0
if messages and messages[0].get("role") == "system":
head = 1
return head + self._effective_protect_first_n()
return head + self._effective_protect_first_n(messages)
def _align_boundary_backward(self, messages: List[Dict[str, Any]], idx: int) -> int:
"""Pull a compress-end boundary backward to avoid splitting a
@ -3593,6 +3671,8 @@ This compaction should PRIORITISE preserving all information related to the focu
msg.get("content")
):
continue
if self._is_context_summary_message(msg):
continue
if last_any < 0:
last_any = i
content = msg.get("content")
@ -4059,23 +4139,45 @@ This compaction should PRIORITISE preserving all information related to the focu
return messages
turns_to_summarize = messages[compress_start:compress_end]
# Snapshot the rehydration state so an aborted attempt below can roll
# it back. The self-heal scan mutates ``_previous_summary`` (populating
# it from a fossil, or discarding a stale cross-session one); if
# summary generation then aborts and returns the transcript unchanged,
# leaving that mutation behind would make the retry — still
# ``compression_count == 0`` but now with a truthy ``_previous_summary``
# — take the narrow rescan, miss a beyond-window fossil, and discard the
# rehydrated state as cross-session leakage (#57835).
_previous_summary_before_scan = self._previous_summary
# A persisted handoff summary can sit in the protected head after a
# resume (commonly immediately after the system prompt). Search from
# the first non-system message through the compression window so we can
# rehydrate iterative-summary state without serializing that handoff as
# a new turn. Protected messages after the handoff remain live context,
# so only summarize messages that are both after the handoff and inside
# the current compression window.
# the first non-system message through the compression window. On the
# first compaction after a restart, extend through the full transcript
# so summaries that landed in the protected tail or drifted past the
# decay probe still rehydrate iterative-summary state instead of being
# copied forward as stacked fossils.
summary_search_start = 1 if messages and messages[0].get("role") == "system" else 0
summary_idx, summary_body = self._find_latest_context_summary(
summary_search_end = compress_end
if self.compression_count < 1 and not self._previous_summary:
summary_search_end = len(messages)
summary_search_end = min(len(messages), summary_search_end)
summary_indices: set[int] = set()
summary_idx = None
summary_body = None
tail_start = compress_end
summary_hits = self._find_context_summaries(
messages,
summary_search_start,
compress_end,
summary_search_end,
)
real_user_present = self._transcript_has_real_user_turn(messages)
if summary_idx is not None:
if summary_body and not self._previous_summary:
self._previous_summary = summary_body
if summary_hits:
summary_idx = summary_hits[-1][0]
summary_body = summary_hits[-1][1]
if not self._previous_summary:
summary_bodies = [body for _, body in summary_hits if body]
if summary_bodies:
self._previous_summary = "\n\n".join(summary_bodies)
# Zero-user provenance (#64650) rides on the newest handoff hit.
provenance = messages[summary_idx].get(
COMPRESSED_SUMMARY_HAS_USER_TURN_KEY
)
@ -4090,7 +4192,19 @@ This compaction should PRIORITISE preserving all information related to the focu
self._summary_has_user_turn = not (
summary_body and _NO_USER_TASK_SENTINEL in summary_body
)
turns_to_summarize = messages[max(compress_start, summary_idx + 1):compress_end]
summary_indices = {idx for idx, _ in summary_hits}
pre_summary_turns = [
msg for idx, msg in enumerate(
messages[compress_start:summary_idx],
start=compress_start,
)
if idx not in summary_indices
]
turns_to_summarize = (
pre_summary_turns + messages[summary_idx + 1:compress_end]
)
if summary_idx >= compress_end:
tail_start = summary_idx + 1
elif self._previous_summary:
# No handoff summary found in the current messages, but
# _previous_summary is non-empty — it was set by a different
@ -4121,7 +4235,7 @@ This compaction should PRIORITISE preserving all information related to the focu
self.threshold_percent * 100,
self.threshold_tokens,
)
tail_msgs = n_messages - compress_end
tail_msgs = n_messages - tail_start
logger.info(
"Summarizing turns %d-%d (%d turns), protecting %d head + %d tail messages",
compress_start + 1,
@ -4174,6 +4288,11 @@ This compaction should PRIORITISE preserving all information related to the focu
telemetry["failure_class"] = "summary_network_failure"
else:
telemetry["failure_class"] = "summary_generation_aborted"
# Roll back the self-heal rehydration so this aborted attempt is a
# true no-op: the next attempt must re-run the full first-compaction
# scan instead of narrow-rescanning against a half-populated state
# and discarding a legitimately rehydrated fossil (#57835).
self._previous_summary = _previous_summary_before_scan
if not self.quiet_mode:
if self._last_summary_auth_failure:
logger.warning(
@ -4214,8 +4333,8 @@ This compaction should PRIORITISE preserving all information related to the focu
# _strip_context_summary_handoff_message() handles both shapes:
# standalone handoffs strip to None (dropped), merged handoffs
# unwrap to their genuine prior-tail content (preserved). Do NOT
# short-circuit on summary_idx here: a merged handoff carries real
# user content that a blanket skip would silently delete.
# short-circuit on summary_indices here: a merged handoff carries
# real user content that a blanket skip would silently delete.
msg = _fresh_compaction_message_copy(messages[i])
if i == 0 and msg.get("role") == "system":
existing = msg.get("content")
@ -4246,17 +4365,26 @@ This compaction should PRIORITISE preserving all information related to the focu
)
tail_messages: List[Dict[str, Any]] = []
for i in range(compress_end, n_messages):
# Start at tail_start (not compress_end): the restart-decay scan may
# have advanced it past a summary that sat beyond compress_end
# (#57835). summary_indices rows are already rehydrated; the strip
# helper handles any that remain (standalone → dropped, merged →
# unwrapped to genuine prior-tail content, #47274).
for i in range(max(compress_end, tail_start), n_messages):
if i in summary_indices and i >= tail_start:
# A summary at/after tail_start was already folded into
# _previous_summary; don't re-emit it verbatim.
continue
msg = _fresh_compaction_message_copy(messages[i])
stripped = self._strip_context_summary_handoff_message(msg)
if stripped is not None:
tail_messages.append(stripped)
_merge_summary_into_tail = False
last_head_role = compressed[-1].get("role", "user") if compressed else "user"
# NOTE: derive the tail's leading role from tail_messages (post
# handoff-strip), not messages[compress_end] — a stripped stale
# last_head_role reads the assembled (post-strip) head; first_tail_role
# reads the assembled (post-strip) tail_messages — a stripped stale
# handoff must not influence alternation-safe role selection.
last_head_role = compressed[-1].get("role", "user") if compressed else "user"
first_tail_role = tail_messages[0].get("role", "user") if tail_messages else None
# When the only protected head message is the system prompt, the
# summary becomes the first *visible* message in the API request
@ -4265,7 +4393,7 @@ This compaction should PRIORITISE preserving all information related to the focu
# Anthropic unconditionally rejects requests whose first message
# is not role=user, so we must pin the summary to "user" and
# prevent the flip logic below from reverting it (#52160).
_force_user_leading = last_head_role == "system"
_force_user_leading = compress_start == 0 or last_head_role == "system"
# Zero-user-turn guard (#58753). The #52160 guard above only fires
# when the system prompt sits *inside* ``messages`` (the gateway
# ``/compress`` path). The main auto-compression path passes the
@ -4299,7 +4427,7 @@ This compaction should PRIORITISE preserving all information related to the focu
summary_role = "assistant"
# If the chosen role collides with the tail AND flipping wouldn't
# collide with the head, flip it.
if first_tail_role and summary_role == first_tail_role:
if first_tail_role is not None and summary_role == first_tail_role:
flipped = "assistant" if summary_role == "user" else "user"
if flipped != last_head_role and not _force_user_leading:
summary_role = flipped
@ -4332,27 +4460,39 @@ This compaction should PRIORITISE preserving all information related to the focu
for tail_idx, msg in enumerate(tail_messages):
if _merge_summary_into_tail and tail_idx == 0:
# Merge the summary into the first tail message, but place
# the END MARKER at the very end so the model sees an
# unambiguous boundary. Old tail content is preserved as
# reference material BEFORE the summary, clearly delimited
# so it is not mistaken for a new message to respond to.
# Uses _append_text_to_content to safely handle both
# string and multimodal-list content types.
# Fixes ghost-message leakage across compaction boundaries
# where old head messages survived verbatim and appeared
# before the summary.
# Merge the summary into the first (post-strip) tail message.
old_content = msg.get("content", "")
suffix = (
"\n\n" + _MERGED_SUMMARY_DELIMITER + "\n\n"
+ summary + "\n\n"
+ _SUMMARY_END_MARKER
)
msg["content"] = _append_text_to_content(
_append_text_to_content(old_content, suffix, prepend=False),
_MERGED_PRIOR_CONTEXT_HEADER + "\n",
prepend=True,
)
if _force_user_leading and summary_role == "user":
# The summary must be part of the first user-visible
# message for Anthropic/Bedrock, but the real tail request
# still has to appear *after* the summary boundary.
prefix = summary + "\n\n" + _SUMMARY_END_MARKER + "\n\n"
msg["content"] = _append_text_to_content(
old_content,
prefix,
prepend=True,
)
else:
# Merge the summary into the first tail message, but place
# the END MARKER at the very end so the model sees an
# unambiguous boundary. Old tail content is preserved as
# reference material BEFORE the summary, clearly delimited
# so it is not mistaken for a new message to respond to.
# Uses _append_text_to_content to safely handle both
# string and multimodal-list content types.
# Fixes ghost-message leakage across compaction boundaries
# where old head messages survived verbatim and appeared
# before the summary.
suffix = (
"\n\n" + _MERGED_SUMMARY_DELIMITER + "\n\n"
+ summary + "\n\n"
+ _SUMMARY_END_MARKER
)
msg["content"] = _append_text_to_content(
_append_text_to_content(old_content, suffix, prepend=False),
_MERGED_PRIOR_CONTEXT_HEADER + "\n",
prepend=True,
)
# Mark the merged message so frontends can identify it as
# containing a compression summary prefix.
msg[COMPRESSED_SUMMARY_METADATA_KEY] = True

View file

@ -3898,6 +3898,35 @@ class TestDoubleCompactionSummaryRole:
f"compatibility, got role={non_system[0]['role']!r}"
)
def test_restart_handoff_without_system_still_starts_with_user(self):
"""When decayed head protection leaves no head, the visible transcript
must still begin with a user role for Anthropic/Bedrock compatibility.
"""
mock_response = MagicMock()
mock_response.choices = [MagicMock()]
mock_response.choices[0].message.content = "summary of resumed turns"
with patch("agent.context_compressor.get_model_context_length", return_value=100000):
c = ContextCompressor(
model="test", quiet_mode=True, protect_first_n=3, protect_last_n=2,
)
msgs = [
{"role": "user", "content": f"{SUMMARY_PREFIX}\nold persisted summary"},
{"role": "assistant", "content": "handoff acknowledged"},
{"role": "user", "content": "new work after restart"},
{"role": "assistant", "content": "new answer after restart"},
{"role": "user", "content": "more new work after restart"},
{"role": "assistant", "content": "more new answer after restart"},
{"role": "user", "content": "tail request"},
{"role": "assistant", "content": "tail answer"},
]
with patch("agent.context_compressor.call_llm", return_value=mock_response):
result = c.compress(msgs)
assert result[0]["role"] == "user"
assert "summary of resumed turns" in (result[0].get("content") or "")
def test_double_compaction_user_tail_merges_into_tail(self):
"""When the summary is forced to role=user (system-only head) and
the first tail message is also user, the summary must merge into

View file

@ -8,16 +8,17 @@ from agent.context_compressor import (
SUMMARY_PREFIX,
_MERGED_PRIOR_CONTEXT_HEADER,
_MERGED_SUMMARY_DELIMITER,
_RESTART_HANDOFF_PROBE_EXTRA_MESSAGES,
_SUMMARY_END_MARKER,
)
def _compressor() -> ContextCompressor:
def _compressor(protect_first_n: int = 1) -> ContextCompressor:
with patch("agent.context_compressor.get_model_context_length", return_value=100000):
return ContextCompressor(
model="test/model",
threshold_percent=0.85,
protect_first_n=1,
protect_first_n=protect_first_n,
protect_last_n=1,
quiet_mode=True,
)
@ -58,6 +59,35 @@ def _messages_with_merged_handoff(summary_body: str, prior_tail: str):
return messages
def _messages_with_default_handoff(summary_body: str):
return [
{"role": "system", "content": "system prompt"},
{"role": "user", "content": "original task before first compaction"},
{"role": "assistant", "content": "original answer before first compaction"},
{"role": "user", "content": "original follow-up before first compaction"},
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{summary_body}"},
{"role": "user", "content": "new user turn after restart"},
{"role": "assistant", "content": "new assistant work after restart"},
{"role": "user", "content": "more new work after restart"},
{"role": "assistant", "content": "latest tail response"},
{"role": "user", "content": "final active request stays in protected tail"},
]
def _messages_with_summary_at_index(summary_index: int):
msgs = [{"role": "system", "content": "system prompt"}]
for idx in range(1, summary_index):
role = "user" if idx % 2 else "assistant"
msgs.append({"role": role, "content": f"probe filler {idx}"})
role = "user" if summary_index % 2 else "assistant"
msgs.append({"role": role, "content": f"{SUMMARY_PREFIX}\nboundary summary"})
msgs.extend([
{"role": "assistant", "content": "new answer"},
{"role": "user", "content": "tail request"},
])
return msgs
def test_existing_previous_summary_is_not_serialized_again_as_new_turn():
"""Same-process iterative compression should not feed the old handoff twice."""
compressor = _compressor()
@ -258,3 +288,367 @@ def test_legacy_multimodal_merged_handoff_preserves_original_blocks():
"role": "user",
"content": [prior_text, prior_image],
}
def test_resume_handoff_in_protected_head_is_not_preserved_as_fossil():
"""After restart, a persisted handoff summary should decay head protection."""
compressor = _compressor()
old_summary = "RESTART-FOSSIL-SUMMARY durable facts from before restart"
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")):
result = compressor.compress(_messages_with_handoff(old_summary))
assert compressor._previous_summary == "fresh summary"
summary_messages = [
msg for msg in result
if ContextCompressor._has_compressed_summary_metadata(msg)
or ContextCompressor._is_context_summary_content(msg.get("content"))
]
assert len(summary_messages) == 1
assert all(
old_summary not in str(msg.get("content", ""))
for msg in result
)
def test_resume_handoff_after_default_protected_head_decays_initial_turns():
"""Default protect_first_n=3 should not fossilize old protected head turns."""
compressor = _compressor(protect_first_n=3)
old_summary = "DEFAULT-RESTART-SUMMARY durable facts from before restart"
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")) as mock_call:
result = compressor.compress(_messages_with_default_handoff(old_summary))
prompt = mock_call.call_args.kwargs["messages"][0]["content"]
assert "PREVIOUS SUMMARY:" in prompt
assert prompt.count(old_summary) == 1
assert "original task before first compaction" in prompt
assert "original answer before first compaction" in prompt
assert "original follow-up before first compaction" in prompt
assert f"[ASSISTANT]: {SUMMARY_PREFIX}" not in prompt
assert compressor._previous_summary == "fresh summary"
assert all(
"original task before first compaction" not in str(msg.get("content", ""))
for msg in result
)
assert all(
"original answer before first compaction" not in str(msg.get("content", ""))
for msg in result
)
assert all(
old_summary not in str(msg.get("content", ""))
for msg in result
)
def test_tail_summary_marker_does_not_decay_first_compaction_head():
"""A live tail summary-looking message should not mimic a resumed handoff."""
compressor = _compressor(protect_first_n=3)
tail_summary = "TAIL-SUMMARY-LIKE message belongs to current protected tail"
msgs = [
{"role": "system", "content": "system prompt"},
{"role": "user", "content": "HEAD-ONE original request"},
{"role": "assistant", "content": "HEAD-TWO original answer"},
{"role": "user", "content": "HEAD-THREE original follow-up"},
{"role": "assistant", "content": "middle answer one"},
{"role": "user", "content": "middle request two"},
{"role": "assistant", "content": "middle answer two"},
{"role": "user", "content": "middle request three"},
{"role": "assistant", "content": "middle answer three"},
{"role": "user", "content": "middle request four"},
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{tail_summary}"},
{"role": "user", "content": "final active request stays in protected tail"},
]
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")):
result = compressor.compress(msgs)
result_text = "\n".join(str(msg.get("content", "")) for msg in result)
assert "HEAD-ONE original request" in result_text
assert "HEAD-TWO original answer" in result_text
assert "HEAD-THREE original follow-up" in result_text
assert tail_summary not in result_text
def test_restart_handoff_in_protected_tail_is_folded_not_preserved():
"""Short resumed transcripts should not copy old summaries as tail."""
compressor = _compressor(protect_first_n=3)
old_summary = "TAIL-PROTECTED-OLD-SUMMARY durable facts"
msgs = [
{"role": "system", "content": "system prompt"},
{"role": "user", "content": "original task"},
{"role": "assistant", "content": "original answer"},
{"role": "user", "content": "original follow-up"},
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{old_summary}"},
{"role": "user", "content": "active request"},
]
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")) as mock_call:
result = compressor.compress(msgs)
prompt = mock_call.call_args.kwargs["messages"][0]["content"]
assert "PREVIOUS SUMMARY:" in prompt
assert prompt.count(old_summary) == 1
result_text = "\n".join(str(msg.get("content", "")) for msg in result)
assert old_summary not in result_text
assert "active request" in result_text
assert sum(
1 for msg in result if ContextCompressor._is_context_summary_message(msg)
) == 1
def test_restart_handoff_fallback_preserves_rehydrated_summary_body():
"""Deterministic fallback should retain the rehydrated old summary."""
compressor = _compressor(protect_first_n=3)
old_summary = "FALLBACK-OLD-SUMMARY durable fact must survive"
with patch.object(compressor, "_generate_summary", return_value=None):
result = compressor.compress(_messages_with_default_handoff(old_summary))
result_text = "\n".join(str(msg.get("content", "")) for msg in result)
assert result_text.count(old_summary) == 1
assert sum(
1 for msg in result if ContextCompressor._is_context_summary_message(msg)
) == 1
def test_zero_protect_first_n_still_folds_restart_fossil():
"""protect_first_n=0 should still self-heal restarted summaries."""
compressor = _compressor(protect_first_n=0)
old_summary = "OLD-SUMMARY-ZERO-PROTECT durable facts"
msgs = [
{"role": "system", "content": "system prompt"},
{"role": "user", "content": "task one"},
{"role": "assistant", "content": "answer one"},
{"role": "user", "content": "task two"},
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{old_summary}"},
{"role": "user", "content": "active request"},
]
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")):
result = compressor.compress(msgs)
result_text = "\n".join(str(msg.get("content", "")) for msg in result)
assert old_summary not in result_text
assert result_text.index(_SUMMARY_END_MARKER) < result_text.index("active request")
assert sum(
1 for msg in result if ContextCompressor._is_context_summary_message(msg)
) == 1
def test_fossil_beyond_restart_probe_window_is_still_folded():
"""Self-heal should find summaries that drift past the decay probe."""
compressor = _compressor(protect_first_n=1)
old_summary = "OLD-SUMMARY-FAR-FROM-HEAD durable facts"
msgs = [{"role": "system", "content": "system prompt"}]
msgs += [
{
"role": "user" if idx % 2 else "assistant",
"content": f"filler {idx}",
}
for idx in range(1, 6)
]
msgs += [
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{old_summary}"},
{"role": "user", "content": "active request"},
]
assert compressor._effective_protect_first_n(msgs) == compressor.protect_first_n
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")):
result = compressor.compress(msgs)
assert all(old_summary not in str(msg.get("content", "")) for msg in result)
def test_restart_fossil_survives_summary_abort_then_retry():
"""An aborted first compaction must not strand the rehydrated fossil.
Regression for the abort/retry path. The first-compaction self-heal scan
(``compression_count < 1``) populates ``_previous_summary`` from a fossil
that drifted past the decay probe. If summary generation then aborts
(auth / network / ``abort_on_summary_failure``) and returns the transcript
unchanged, the aborted attempt must not leave that rehydrated state behind:
otherwise the retry still ``compression_count == 0`` but now with a
truthy ``_previous_summary`` takes the narrow rescan, misses the
beyond-window fossil, and then discards the rehydrated summary as
cross-session leakage, copying the fossil forward as a stacked summary.
"""
compressor = _compressor(protect_first_n=1)
compressor.abort_on_summary_failure = True
old_summary = "ABORT-RETRY-OLD-SUMMARY durable facts"
msgs = [{"role": "system", "content": "system prompt"}]
msgs += [
{
"role": "user" if idx % 2 else "assistant",
"content": f"filler {idx}",
}
for idx in range(1, 6)
]
msgs += [
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{old_summary}"},
{"role": "user", "content": "active request"},
]
# First compaction aborts on a summary-generation failure. The transcript
# is returned unchanged AND the self-heal state it rehydrated must be
# rolled back, so a retry behaves like the original first compaction.
with patch.object(compressor, "_generate_summary", return_value=None):
aborted = compressor.compress([dict(m) for m in msgs])
assert compressor._last_compress_aborted is True
assert all(m["content"] for m in aborted) # returned unchanged
assert compressor.compression_count == 0
assert compressor._previous_summary is None
# Retry: the fossil beyond the narrow window is still folded, not copied
# forward as a second stacked summary.
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")):
result = compressor.compress([dict(m) for m in msgs])
assert all(old_summary not in str(msg.get("content", "")) for msg in result)
assert sum(
1 for msg in result if ContextCompressor._is_context_summary_message(msg)
) == 1
def test_tail_turns_before_late_handoff_are_not_lost():
"""Live tail turns before a late handoff should be summarized or kept."""
compressor = _compressor(protect_first_n=3)
old_summary = "LATE-TAIL-OLD-SUMMARY"
msgs = [{"role": "system", "content": "system prompt"}]
msgs += [
{
"role": "user" if idx % 2 else "assistant",
"content": f"body {idx}",
}
for idx in range(1, 9)
]
msgs += [
{"role": "assistant", "content": "TAIL-BEFORE-SUMMARY-A"},
{"role": "user", "content": "TAIL-BEFORE-SUMMARY-B"},
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{old_summary}"},
{"role": "user", "content": "final active request"},
]
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")) as mock_call:
result = compressor.compress(msgs)
preserved = mock_call.call_args.kwargs["messages"][0]["content"] + "\n" + "\n".join(
str(msg.get("content", "")) for msg in result
)
assert "TAIL-BEFORE-SUMMARY-A" in preserved
assert "TAIL-BEFORE-SUMMARY-B" in preserved
def test_forced_leading_merged_summary_strips_live_tail_from_summary_body():
"""Rehydrating a forced-leading merged summary should ignore live tail."""
merged = (
f"{SUMMARY_PREFIX}\nSUMMARY_BODY\n\n"
f"{_SUMMARY_END_MARKER}\n\n"
"LIVE_TAIL_REQUEST"
)
assert ContextCompressor._is_context_summary_content(merged) is True
assert ContextCompressor._strip_summary_prefix(merged) == "SUMMARY_BODY"
def test_restart_probe_boundary_summary_just_inside_window_decays():
"""A summary at the last restart-probe index should still decay."""
compressor = _compressor(protect_first_n=3)
first_non_system = 1
last_probe_idx = (
first_non_system
+ compressor.protect_first_n
+ _RESTART_HANDOFF_PROBE_EXTRA_MESSAGES
- 1
)
assert (
compressor._effective_protect_first_n(
_messages_with_summary_at_index(last_probe_idx)
)
== 0
)
def test_restart_probe_boundary_summary_just_outside_window_does_not_decay():
"""A summary past the restart-probe window should not decay."""
compressor = _compressor(protect_first_n=3)
first_non_system = 1
first_outside_probe_idx = (
first_non_system
+ compressor.protect_first_n
+ _RESTART_HANDOFF_PROBE_EXTRA_MESSAGES
)
assert (
compressor._effective_protect_first_n(
_messages_with_summary_at_index(first_outside_probe_idx)
)
== compressor.protect_first_n
)
def test_restart_stacked_handoffs_fold_stray_head_and_collapse_to_single_summary():
"""Stacked restart summaries should keep stray head turns as new input."""
compressor = _compressor(protect_first_n=3)
old_summary = "OLD-ONLY facts from the first compaction"
newer_summary = "NEW-ONLY facts from work after restart"
msgs = [
{"role": "system", "content": "system prompt"},
{"role": "user", "content": "FOSSIL-HEAD-TURN live detail before summary"},
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\n{old_summary}"},
{"role": "user", "content": f"{SUMMARY_PREFIX}\n{newer_summary}"},
{"role": "assistant", "content": "work after restart"},
{"role": "user", "content": "more work after restart"},
{"role": "assistant", "content": "tail answer"},
{"role": "user", "content": "active tail request"},
]
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")) as mock_call:
result = compressor.compress(msgs)
prompt = mock_call.call_args.kwargs["messages"][0]["content"]
assert "PREVIOUS SUMMARY:" in prompt
assert prompt.count(old_summary) == 1
assert prompt.count(newer_summary) == 1
assert "FOSSIL-HEAD-TURN live detail before summary" in prompt
assert f"[ASSISTANT]: {SUMMARY_PREFIX}" not in prompt
assert f"[USER]: {SUMMARY_PREFIX}" not in prompt
summary_messages = [
msg for msg in result
if ContextCompressor._is_context_summary_message(msg)
]
assert len(summary_messages) == 1
assert all(old_summary not in str(msg.get("content", "")) for msg in result)
assert all(newer_summary not in str(msg.get("content", "")) for msg in result)
assert all("FOSSIL-HEAD-TURN" not in str(msg.get("content", "")) for msg in result)
def test_metadata_summary_decay_also_rehydrates_previous_summary():
"""Metadata-only in-process summaries should decay and rehydrate together."""
compressor = _compressor(protect_first_n=3)
msgs = [
{"role": "system", "content": "system prompt"},
{
"role": "assistant",
"content": "metadata-only prior summary",
COMPRESSED_SUMMARY_METADATA_KEY: True,
},
{"role": "user", "content": "new work"},
{"role": "assistant", "content": "new answer"},
{"role": "user", "content": "tail request"},
{"role": "assistant", "content": "tail answer"},
]
with patch("agent.context_compressor.call_llm", return_value=_response("fresh summary")) as mock_call:
compressor.compress(msgs)
prompt = mock_call.call_args.kwargs["messages"][0]["content"]
assert "PREVIOUS SUMMARY:" in prompt
assert "metadata-only prior summary" in prompt
assert compressor._previous_summary == "fresh summary"