mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-20 15:33:54 +00:00
fix(background_review): inherit parent's reasoning_config to preserve Anthropic cache namespace
PR #17276 painstakingly pinned `_cached_system_prompt`, `session_start`, `session_id`, and the toolset config on the background-review fork so its outbound request body would byte-match the parent's and hit Anthropic's exact-prefix cache. The contributor measured a ~26% end-to-end cost reduction on Sonnet 4.5. That optimization is currently being silently undone by a missing `reasoning_config` kwarg. The fork's `AIAgent(...)` call omits it, so the fork's `reasoning_config` defaults to `None`. `anthropic_adapter.build_anthropic_kwargs` (line ~2165) then short-circuits the `thinking` / `output_config` block, and the fork's request body lands in a DIFFERENT Anthropic cache namespace from the parent's. Result on the wire: 0 `cache_read_input_tokens`, full `cache_creation_input_tokens` of the entire parent prefix — every single background review. 7 days of midagent.db traffic from one host running stock Hermes against Anthropic Sonnet: ``` Background-review FIRST calls (the moment a review fork is born): count = 68 cache_write tokens = 7,004,297 cache_read tokens = 1,016,335 Cost on Sonnet ($3.75/M write vs $0.30/M read): Spent on these writes: $26.27 Cost if they had hit parent cache instead: $2.10 WASTED: $24.16 / week / user ``` That is from one user. Multiply by Hermes's installed base for the full impact. Tested against api.anthropic.com directly (see refs/api-tests/ in the attached investigation repo if needed): | pair | cache_r | cache_w | |---------------------------------------------|---------|---------| | parent fresh | 0 | 24,047 | | parent same again | 24,047 | 0 | | fork: appends 2 new tail msgs, thinking ON | 24,047 | 22 | | fork: appends 2 new tail msgs, thinking OFF | 0 | 24,047 | Same fork-shape request, only difference is `thinking`. With the fix, the fork hits the parent's full prefix and only writes the delta (the `Review the conversation above…` prompt block, ~3-5K tokens). One line in `agent/background_review.py`: pass `reasoning_config=getattr(agent, "reasoning_config", None)` to the `AIAgent(...)` constructor of the review fork. A short comment block above it explains why so the next person who reads this code doesn't re-introduce the regression. `tests/run_agent/test_background_review_cache_parity.py` already covers the system-prompt / session-id / toolset-config parity contracts that PR #17276 introduced. I added: * a `reasoning_config` attribute to `_make_agent_stub` so the stub has a non-None parent value the test can verify is propagated. * `test_review_fork_inherits_parent_reasoning_config()` — asserts the fork's `AIAgent(...)` kwargs carry the parent's `reasoning_config`. Pre-fix this test fails with `None vs expected {'enabled': True, 'effort': 'medium'}`; post-fix all 4 tests in the file pass. ``` $ python -m pytest tests/run_agent/test_background_review_cache_parity.py -v test_review_fork_inherits_parent_cached_system_prompt PASSED test_review_fork_pins_session_start_and_session_id PASSED test_review_fork_inherits_parent_toolset_config PASSED test_review_fork_inherits_parent_reasoning_config PASSED ← new ``` Also runs against the broader background-review test suite: `test_background_review.py` (4), `test_background_review_summary.py` (8), `test_background_review_toolset_restriction.py` (3) — 19/19 pass. `agent/curator.py:1691` has the same omission for the umbrella-curation fork, but curator's prompt is "curate all skills" — it shares no prefix with any user conversation, so cache-parity is a non-issue there. Worth auditing if the curator ever takes a parent conversation as input, but not part of this PR. The `agent/auxiliary_client.py:1006` `reasoning_config=None` hardcode is intentional (title/summary one-shots on short prompts — per-call cost of namespace flip is negligible) and is also out of scope.
This commit is contained in:
parent
ca559a7852
commit
17cfa0f0a5
2 changed files with 53 additions and 0 deletions
|
|
@ -696,6 +696,10 @@ def _run_review_in_thread(
|
|||
if isinstance(_rt.get("command"), str) and _rt["command"]:
|
||||
_fork_kwargs["acp_command"] = _rt["command"]
|
||||
_fork_kwargs["acp_args"] = _rt.get("args") or []
|
||||
# Match parent's reasoning config so the fork's ``thinking`` /
|
||||
# ``output_config`` are byte-identical in the request body —
|
||||
# Anthropic's cache key is namespaced by ``thinking`` presence.
|
||||
_fork_kwargs["reasoning_config"] = getattr(agent, "reasoning_config", None)
|
||||
review_agent = AIAgent(
|
||||
model=_rt.get("model") or agent.model,
|
||||
max_iterations=16,
|
||||
|
|
|
|||
|
|
@ -41,6 +41,9 @@ def _make_agent_stub(agent_cls):
|
|||
# Non-None so the test catches a missing-kwarg regression.
|
||||
agent.enabled_toolsets = ["memory", "skills", "terminal"]
|
||||
agent.disabled_toolsets = ["spotify", "feishu_doc"]
|
||||
# Non-None so the test catches reasoning_config NOT being inherited —
|
||||
# which would put the fork into a different Anthropic cache namespace.
|
||||
agent.reasoning_config = {"enabled": True, "effort": "medium"}
|
||||
return agent
|
||||
|
||||
|
||||
|
|
@ -237,3 +240,49 @@ def test_review_fork_inherits_parent_toolset_config():
|
|||
f"disabled_toolsets mismatch: {captured.get('disabled_toolsets')!r} "
|
||||
f"vs expected {agent.disabled_toolsets!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_review_fork_inherits_parent_reasoning_config():
|
||||
"""``reasoning_config`` parity: fork must inherit parent's value so the request body's ``thinking`` / ``output_config`` match (Anthropic cache is namespaced by ``thinking`` presence)."""
|
||||
import run_agent
|
||||
|
||||
agent = _make_agent_stub(run_agent.AIAgent)
|
||||
|
||||
captured = {}
|
||||
|
||||
class _Recorder:
|
||||
def __init__(self, *args, **kwargs):
|
||||
captured["reasoning_config"] = kwargs.get("reasoning_config")
|
||||
self._cached_system_prompt = None
|
||||
self._memory_write_origin = None
|
||||
self._memory_write_context = None
|
||||
self._memory_store = None
|
||||
self._memory_enabled = None
|
||||
self._user_profile_enabled = None
|
||||
self._memory_nudge_interval = None
|
||||
self._skill_nudge_interval = None
|
||||
self.suppress_status_output = None
|
||||
self.session_start = None
|
||||
self.session_id = None
|
||||
|
||||
def run_conversation(self, *args, **kwargs):
|
||||
raise RuntimeError("stop after recording — don't actually call the API")
|
||||
|
||||
def shutdown_memory_provider(self):
|
||||
pass
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
with patch.object(run_agent, "AIAgent", _Recorder), \
|
||||
patch("threading.Thread", _SyncThread):
|
||||
agent._spawn_background_review(
|
||||
messages_snapshot=[],
|
||||
review_memory=True,
|
||||
review_skills=False,
|
||||
)
|
||||
|
||||
assert captured.get("reasoning_config") == agent.reasoning_config, (
|
||||
f"reasoning_config mismatch: {captured.get('reasoning_config')!r} "
|
||||
f"vs expected {agent.reasoning_config!r}"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue