hermes-agent/tests/agent/test_micro_compaction.py
Michael Jordan cac9526d2a feat(agent): token telemetry for micro-compaction
The existing log line reports message counts, which is the least
informative number available here: absorbing one tool-heavy exchange can
drop hundreds of tokens while moving the count by one. There was no way to
answer "is this actually helping?" from a real session.

Emit one content-free JSON line per pass, in the same shape as the batch
compaction telemetry: before/after tokens, the delta, the size of the
absorbed exchange, the rolling summary size, duration, and running
per-session totals so a whole run can be read off the last line. No
transcript content rides along.

Add scripts/micro_compaction_report.py to aggregate those lines into
passes, outcome mix, net tokens saved, mean exchange size and durations,
with an optional per-session breakdown.

Measuring it immediately surfaced something worth documenting: the first
pass in a session normally *costs* tokens. The summary marker carries a
fixed ~400 tokens of scaffolding, paid on pass one against a single
absorbed exchange. From pass two the marker is replaced rather than added,
so the overhead is already paid and each exchange is close to pure saving.
Break-even is typically the second or third pass. Tests cover the
telemetry contract, the cumulative totals, and that first-pass/later-pass
shape so nobody reads a single turn and concludes it made things worse.

The estimator costs ~5 ms at 600 messages and ~20 ms at 1200, taken twice
per pass, post-turn — and only once an exchange is actually in hand, so
turns that no-op early pay nothing.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-31 17:44:19 +05:30

269 lines
10 KiB
Python

"""Tests for per-turn micro-compaction in ``ContextCompressor``.
Micro-compaction amortizes the cost of context compression: instead of one
long pause when the window fills, each turn folds the single oldest
un-absorbed exchange into a rolling summary.
The invariants that matter:
* one call absorbs exactly one exchange (assistant + its tool results), so
the per-turn cost stays bounded;
* the absorbed span is replaced by a summary marker carrying the usual
``_compressed_summary`` metadata, so resume/handoff treat it like a batch
summary;
* the cursor advances, so successive calls walk forward rather than
re-summarising the same exchange;
* protected head and tail messages are never touched;
* an exchange the summarizer cannot handle is retried a bounded number of
times and then skipped, so a poison exchange can't stall every turn.
"""
import pytest
from agent.context_compressor import (
COMPRESSED_SUMMARY_METADATA_KEY,
ContextCompressor,
_MICRO_COMPACT_MAX_CONSECUTIVE_FAILURES,
)
def _compressor(summary="ROLLING SUMMARY") -> ContextCompressor:
cc = ContextCompressor(
model="test-model",
threshold_percent=0.75,
protect_first_n=1,
protect_last_n=2,
quiet_mode=True,
config_context_length=40960,
provider="test",
)
cc._micro_compact_enabled = True
# Stand in for the auxiliary summarizer LLM.
cc._micro_summarize_one = lambda _text: summary
return cc
def _conversation(exchanges: int = 6) -> list:
msgs = [{"role": "system", "content": "system prompt"}]
for i in range(exchanges):
msgs.append({"role": "user", "content": f"question {i}"})
msgs.append({"role": "assistant", "content": f"answer {i} " + "z" * 400})
return msgs
def _summary_markers(messages: list) -> list:
return [m for m in messages if m.get(COMPRESSED_SUMMARY_METADATA_KEY)]
class TestMicroCompaction:
def test_absorbs_one_exchange_and_leaves_a_summary_marker(self):
cc = _compressor()
messages = _conversation()
result = cc._micro_compact(list(messages))
# The absorbed assistant turn is gone from the transcript.
assert any("answer 0" in str(m.get("content")) for m in messages)
assert not any("answer 0" in str(m.get("content")) for m in result)
markers = _summary_markers(result)
assert len(markers) == 1
assert "ROLLING SUMMARY" in markers[0]["content"]
# The marker stands in for a user turn, like batch compaction's does.
assert markers[0]["role"] == "user"
def test_disabled_is_a_no_op(self):
cc = _compressor()
cc._micro_compact_enabled = False
messages = _conversation()
assert cc._micro_compact(list(messages)) == messages
def test_cursor_advances_across_successive_turns(self):
cc = _compressor()
messages = _conversation(exchanges=8)
first = cc._micro_compact(list(messages))
cursor_after_first = cc._micro_compact_cursor
second = cc._micro_compact(list(first))
assert cursor_after_first > 0
assert cc._micro_compact_cursor >= cursor_after_first
# Still exactly one marker: the second pass merges into the rolling
# summary rather than stacking a second summary block.
assert len(_summary_markers(second)) == 1
def test_protected_head_and_tail_survive(self):
cc = _compressor()
messages = _conversation()
result = cc._micro_compact(list(messages))
assert result[0] == messages[0], "system prompt must be preserved"
assert result[-1] == messages[-1], "most recent turn must be preserved"
def test_user_messages_are_never_absorbed(self):
"""User turns stay verbatim for the life of the session — by design.
Assistant output is largely an account of what was done and survives
summarising; the user's own words are the intent everything else is
derived from and can't be reconstructed from it. So an exchange starts
at the assistant message and the walk skips past user turns.
"""
cc = _compressor()
messages = _conversation(exchanges=10)
originals = [m["content"] for m in messages if m["role"] == "user"]
for _ in range(5):
messages = cc._micro_compact(messages)
surviving = [
m["content"] for m in messages
if m.get("role") == "user" and not m.get(COMPRESSED_SUMMARY_METADATA_KEY)
]
assert surviving == originals, "user turns must survive verbatim"
def test_short_conversation_is_untouched(self):
cc = _compressor()
messages = [
{"role": "system", "content": "sys"},
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "hello"},
]
assert cc._micro_compact(list(messages)) == messages
def test_summarizer_failure_leaves_conversation_intact(self):
cc = _compressor()
cc._micro_summarize_one = lambda _text: None
messages = _conversation()
result = cc._micro_compact(list(messages))
assert result == messages
assert cc._micro_compact_consecutive_failures == 1
def test_poison_exchange_is_skipped_after_repeated_failures(self):
"""A repeatedly unsummarizable exchange must not stall every turn."""
cc = _compressor()
cc._micro_summarize_one = lambda _text: None
messages = _conversation()
for _ in range(_MICRO_COMPACT_MAX_CONSECUTIVE_FAILURES):
cc._micro_compact(list(messages))
# The cursor has moved past the stuck exchange and the strike count
# is reset, so the next turn attempts new material.
assert cc._micro_compact_cursor > 0
assert cc._micro_compact_consecutive_failures == 0
def test_repeated_compaction_shrinks_context_and_keeps_one_marker(self):
"""The whole point: successive turns must reduce the transcript.
The rolling summary is cumulative, so an earlier marker's text is a
subset of the current one. Keeping the earlier markers stacked
near-duplicate copies (each with its own heading/end-marker
scaffolding) and made the transcript grow every turn — the opposite
of what compaction is for.
"""
from agent.model_metadata import estimate_messages_tokens_rough
cc = _compressor()
# Cumulative summary, like the real summarizer produces.
state = {"n": 0}
def growing(_text):
state["n"] += 1
return "SUMMARY " + " ".join(f"ex{i}" for i in range(state["n"]))
cc._micro_summarize_one = growing
messages = _conversation(exchanges=12)
before = estimate_messages_tokens_rough(messages)
for _ in range(6):
messages = cc._micro_compact(messages)
after = estimate_messages_tokens_rough(messages)
assert len(_summary_markers(messages)) == 1
assert after < before, f"context grew: {before} -> {after}"
def test_emits_content_free_token_telemetry(self, caplog):
"""Each pass logs one JSON line with the token accounting.
Message counts barely move even when the saving is large, so the token
fields are what make the effect measurable in a real session.
"""
import json
import logging
cc = _compressor()
messages = _conversation(exchanges=8)
with caplog.at_level(logging.INFO, logger="agent.context_compressor"):
result = cc._micro_compact(messages)
lines = [
r.getMessage() for r in caplog.records
if "micro compaction telemetry:" in r.getMessage()
]
assert len(lines) == 1
payload = json.loads(lines[0].split("micro compaction telemetry: ", 1)[1])
assert payload["event"] == "micro_compaction"
assert payload["outcome"] == "absorbed"
assert payload["tokens_saved_total"] == -payload["tokens_delta"]
assert payload["passes_total"] == 1
assert payload["messages_after"] == len(result)
assert payload["exchange_tokens"] > 0
# Content-free: no transcript text may ride along in the payload.
blob = json.dumps(payload)
assert "answer 0" not in blob and "question 0" not in blob
def test_first_pass_costs_marker_overhead_then_pays_it_back(self):
"""The first pass can grow the transcript; later passes recover it.
Inserting the summary marker costs a fixed ~400 tokens of scaffolding
(the compaction preamble, the historical heading and the end marker).
On pass one that overhead is paid against a single absorbed exchange,
so the net can be positive. From pass two on the marker is replaced
rather than added, so the scaffolding is already paid for and each
absorbed exchange is pure saving. Anyone reading a single turn's
telemetry needs to know this before concluding it made things worse.
"""
from agent.model_metadata import estimate_messages_tokens_rough
cc = _compressor()
messages = _conversation(exchanges=10)
start = estimate_messages_tokens_rough(messages)
messages = cc._micro_compact(messages)
after_first = estimate_messages_tokens_rough(messages)
for _ in range(5):
messages = cc._micro_compact(messages)
after_many = estimate_messages_tokens_rough(messages)
assert after_first > start, "expected one-time marker overhead"
assert after_many < after_first, "later passes must recover it"
def test_cumulative_savings_accumulate_across_passes(self):
cc = _compressor()
messages = _conversation(exchanges=10)
for _ in range(4):
messages = cc._micro_compact(messages)
assert cc._micro_compact_passes == 4
assert cc._micro_compact_tokens_saved_total > 0
def test_defrag_triggers_once_the_rolling_summary_grows(self):
cc = _compressor(summary="FRESH DEFRAGGED SUMMARY")
cc._micro_compact_rolling_summary = "x" * 40_000 # far over the threshold
messages = _conversation(exchanges=8)
assert cc._needs_defrag() is True
result = cc._micro_compact(list(messages))
assert cc._micro_compact_rolling_summary == "FRESH DEFRAGGED SUMMARY"
markers = _summary_markers(result)
assert len(markers) == 1
assert "FRESH DEFRAGGED SUMMARY" in markers[0]["content"]