hermes-agent/tests/agent/test_cjk_token_estimation.py
Teknium ea0fd393db perf(compression): gate CJK-aware token estimation behind an ASCII fast path
The salvaged estimator ran a per-character Python loop on every
estimate_tokens_rough() call — a ~28,000,000x slowdown vs (len+3)//4 on a
1MB ASCII tool output (measured ~3.0s per call). Gate it:

- str.isascii() O(1) fast path keeps pure-ASCII text bit-identical to the
  classic (len+3)//4 rule at ~1.3x baseline cost (0.23us vs 0.17us per
  1MB call).
- Non-ASCII text counts dense CJK chars via a compiled character-class
  regex in C (len(text) - len(re.sub(''))): ~352ms/1MB hangul vs ~2.1s
  for the per-char loop.
- Non-ASCII-but-non-CJK text (accents, Cyrillic, emoji) keeps the classic
  rule.

Also: parity tests against the per-char reference implementation, and
updated two stale expectations that encoded the old behavior (CJK now
counted ~1 token/char; short string content now ceil-divided instead of
floored to 0). The continuity test now detects merged-into-tail summaries
via _is_context_summary_content.
2026-07-22 06:57:22 -07:00

94 lines
3.1 KiB
Python

from unittest.mock import patch
from agent.context_compressor import ContextCompressor, _estimate_msg_budget_tokens
from agent.model_metadata import (
_is_cjk_token_dense_char,
estimate_messages_tokens_rough,
estimate_tokens_rough,
)
def test_cjk_text_is_not_estimated_as_four_chars_per_token():
assert estimate_tokens_rough("a" * 400) == 100
assert estimate_tokens_rough("" * 400) >= 400
def test_message_estimate_counts_korean_content_as_token_dense():
messages = [{"role": "user", "content": "압축 테스트 " + ("" * 1000)}]
assert estimate_messages_tokens_rough(messages) >= 1000
def test_compressor_tail_budget_uses_cjk_aware_message_estimate():
korean_msg = {"role": "assistant", "content": "" * 2000}
english_msg = {"role": "assistant", "content": "a" * 2000}
assert _estimate_msg_budget_tokens(korean_msg) > _estimate_msg_budget_tokens(english_msg)
def test_cjk_tail_does_not_expand_to_english_char_budget():
with patch("agent.context_compressor.get_model_context_length", return_value=65536):
compressor = ContextCompressor(
"test/model",
protect_first_n=3,
protect_last_n=20,
summary_target_ratio=0.2,
quiet_mode=True,
)
messages = [
{"role": "user", "content": "head 1"},
{"role": "assistant", "content": "head 2"},
{"role": "user", "content": "head 3"},
]
for idx in range(40):
role = "assistant" if idx % 2 else "user"
messages.append({"role": role, "content": "" * 1200})
compress_start = compressor._align_boundary_forward(
messages,
compressor._protect_head_size(messages),
)
compress_end = compressor._find_tail_cut_by_tokens(messages, compress_start)
assert len(messages) - compress_end < 31
def _reference_per_char_estimate(text: str) -> int:
"""The pre-perf-gate per-character reference implementation."""
dense = 0
sparse = 0
for ch in text:
if _is_cjk_token_dense_char(ch):
dense += 1
else:
sparse += 1
return dense + ((sparse + 3) // 4)
def test_perf_gated_estimator_matches_per_char_reference():
samples = [
"",
"ab",
"a" * 400,
"" * 400,
"압축 테스트 " + ("" * 1000),
"café résumé naïve", # non-ASCII, no CJK
"hello 안녕 world",
"アイウエオ テスト", # halfwidth kana (fullwidth-forms block)
"漢字とかな交じり文です。",
"русский текст", # Cyrillic — non-ASCII, non-CJK
]
for text in samples:
assert estimate_tokens_rough(text) == _reference_per_char_estimate(text), repr(text)
def test_ascii_fast_path_keeps_classic_four_chars_per_token():
# Pure ASCII must be bit-identical to the historical (len+3)//4 rule.
for text in ("x", "xyz", "a" * 1000, "tool output\n" * 500):
assert estimate_tokens_rough(text) == (len(text) + 3) // 4
def test_non_ascii_non_cjk_keeps_classic_rule():
text = "café résumé " * 40
assert estimate_tokens_rough(text) == (len(text) + 3) // 4