hermes-agent/tests/agent/test_cjk_token_estimation.py
Teknium 6b81590c55
test: prune low-value tests suite-wide (wave 1) — 46,820 → 28,106 test functions
Systematic prune per AGENTS.md test policy, one pass over every major
test tree (gateway, hermes_cli, tools, agent, run_agent, plugins, cli,
cron, tui_gateway, honcho/openviking, root-level):

- DELETE: source-reading tests (read_text/getsource on prod files),
  change-detector tests (exact catalog counts, model-name snapshots,
  config version literals), mock-echo tests (assert a mock returns what
  it was told), assertion-free/trivial tests, near-duplicate
  parametrizations (boundaries + one representative kept), async/sync
  twin duplicates, cosmetic within-file variations.
- KEEP (mandatory): security/redaction/approval guards, message-role
  alternation invariants, prompt-caching/deterministic-call-id
  invariants, issue-number regression tests (deduped), E2E tests.
- 6 test files deleted outright (script-style/no-assert or fully
  redundant); conftest.py, fakes/, fixtures/ untouched.
- tests/acp/conftest.py added: autouse fixture stubs the live
  models.dev/GitHub/Copilot/Anthropic inventory fetches that ACP server
  tests performed on every session create — test_server.py 147s → 3.4s,
  and the tests are now genuinely hermetic.
- Sleep-based slowness shrunk where safe (codex_ttfb_watchdog,
  compression_concurrent_fork, etc.); no wall-clock assertion tightened.

Verification: full hermetic suite via scripts/run_tests.sh —
2439 files, 31,130 tests passed, 0 failed, 0 flaky retries, 315s wall
(baseline: 583s wall, 13,564s subprocess CPU).
2026-07-29 13:10:23 -07:00

81 lines
2.3 KiB
Python

from unittest.mock import patch
from agent.context_compressor import ContextCompressor, _estimate_msg_budget_tokens
from agent.model_metadata import (
_is_cjk_token_dense_char,
estimate_messages_tokens_rough,
estimate_tokens_rough,
)
def test_message_estimate_counts_korean_content_as_token_dense():
messages = [{"role": "user", "content": "압축 테스트 " + ("" * 1000)}]
assert estimate_messages_tokens_rough(messages) >= 1000
def test_cjk_tail_does_not_expand_to_english_char_budget():
with patch("agent.context_compressor.get_model_context_length", return_value=65536):
compressor = ContextCompressor(
"test/model",
protect_first_n=3,
protect_last_n=20,
summary_target_ratio=0.2,
quiet_mode=True,
)
# Resolve while the mock is active (lazy init, #32221).
_ = compressor.context_length
messages = [
{"role": "user", "content": "head 1"},
{"role": "assistant", "content": "head 2"},
{"role": "user", "content": "head 3"},
]
for idx in range(40):
role = "assistant" if idx % 2 else "user"
messages.append({"role": role, "content": "" * 1200})
compress_start = compressor._align_boundary_forward(
messages,
compressor._protect_head_size(messages),
)
compress_end = compressor._find_tail_cut_by_tokens(messages, compress_start)
assert len(messages) - compress_end < 31
def _reference_per_char_estimate(text: str) -> int:
"""The pre-perf-gate per-character reference implementation."""
dense = 0
sparse = 0
for ch in text:
if _is_cjk_token_dense_char(ch):
dense += 1
else:
sparse += 1
return dense + ((sparse + 3) // 4)
def test_perf_gated_estimator_matches_per_char_reference():
samples = [
"",
"ab",
"a" * 400,
"" * 400,
"압축 테스트 " + ("" * 1000),
"café résumé naïve", # non-ASCII, no CJK
"hello 안녕 world",
"アイウエオ テスト", # halfwidth kana (fullwidth-forms block)
"漢字とかな交じり文です。",
"русский текст", # Cyrillic — non-ASCII, non-CJK
]
for text in samples:
assert estimate_tokens_rough(text) == _reference_per_char_estimate(text), repr(text)