From 958a81a1e4386780f544733ff019e7739df64d1d Mon Sep 17 00:00:00 2001 From: Rod Boev Date: Thu, 4 Jun 2026 08:09:04 -0400 Subject: [PATCH] perf(agent): defer synchronous httpx.post out of AIAgent.__init__ (#32221) (cherry picked from commit 1fe63238b012e4326c975b2b1a303f6f27aa5102) --- agent/context_compressor.py | 124 +++++++++++++++++-------- tests/agent/test_context_compressor.py | 65 +++++++++++++ 2 files changed, 152 insertions(+), 37 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 73fa1e36f215..38ed9b4df3ae 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1354,6 +1354,79 @@ class ContextCompressor(ContextEngine): previous = telemetry.get("aux_call_duration_ms") or 0 telemetry["aux_call_duration_ms"] = previous + max(0, int(duration_ms)) + def _resolve_context_length(self) -> int: + """Resolve and cache the model's context length on first access.""" + if self._resolved_context_length is None: + self._resolved_context_length = get_model_context_length( + self.model, + base_url=self.base_url, + api_key=self.api_key, + config_context_length=self._config_context_length, + provider=self.provider, + ) + # Small-context threshold floor: models under 512K trigger at + # >=75% so compaction doesn't fire with half the window still + # free. Raise-only; must run AFTER context_length is resolved + # and BEFORE threshold_tokens is derived (deferred here from + # __init__ along with the resolution itself, #32221). + # _base_threshold_percent already has the per-model override + # applied, so the floor stacks on top of it. + self.threshold_percent = self._effective_threshold_percent( + self._resolved_context_length, self._base_threshold_percent, + ) + return self._resolved_context_length + + @property + def context_length(self) -> int: + return self._resolve_context_length() + + @context_length.setter + def context_length(self, value: int) -> None: + self._resolved_context_length = value + self._threshold_tokens = None + self._tail_token_budget = None + self._max_summary_tokens = None + + @property + def threshold_tokens(self) -> int: + if self._threshold_tokens is None: + # Floor: never compress below MINIMUM_CONTEXT_LENGTH tokens even + # if the percentage would suggest a lower value (#14690 handles + # the degenerate small-window case inside the helper). + self._threshold_tokens = self._compute_threshold_tokens( + self.context_length, self.threshold_percent, self.max_tokens, + ) + # Apply absolute token cap (compression.threshold_tokens) — + # takes the lower of the ratio-based threshold and the cap. + self._apply_threshold_tokens_cap() + return self._threshold_tokens + + @threshold_tokens.setter + def threshold_tokens(self, value: int) -> None: + self._threshold_tokens = value + + @property + def tail_token_budget(self) -> int: + if self._tail_token_budget is None: + self._tail_token_budget = int(self.threshold_tokens * self.summary_target_ratio) + return self._tail_token_budget + + @tail_token_budget.setter + def tail_token_budget(self, value: int) -> None: + self._tail_token_budget = value + + @property + def max_summary_tokens(self) -> int: + if self._max_summary_tokens is None: + self._max_summary_tokens = min( + int(self.context_length * 0.05), _SUMMARY_TOKENS_CEILING, + ) + return self._max_summary_tokens + + @max_summary_tokens.setter + def max_summary_tokens(self, value: int) -> None: + self._max_summary_tokens = value + def on_session_end(self, session_id: str, messages: List[Dict[str, Any]]) -> None: """Clear all per-session compaction state at a real session boundary. @@ -1988,46 +2061,23 @@ class ContextCompressor(ContextEngine): # deterministic "summary unavailable" handoff and drop the middle window. self.abort_on_summary_failure = abort_on_summary_failure - self.context_length = get_model_context_length( - model, base_url=base_url, api_key=api_key, - config_context_length=config_context_length, - provider=provider, - ) - # Small-context threshold floor: models under 512K trigger at >=75% - # so compaction doesn't fire with half the window still free (the - # incompressible floor makes 50%-triggered compaction thrash on - # 128K-262K models). Raise-only; must run AFTER context_length is - # resolved and BEFORE threshold_tokens is derived. The pre-floor - # value is kept so update_model() can re-derive for a new window - # (switching small -> large must drop back to the configured value). - # Note: _base_threshold_percent already has the per-model override - # applied, so the floor stacks on top of any model-specific threshold. + # Defer context-length resolution to first access (#32221): + # get_model_context_length() can issue a synchronous /models HTTP + # probe, which must not block AIAgent construction. The small-context + # threshold floor and the absolute threshold cap both need the + # resolved window, so they are applied on first resolution (see + # _resolve_context_length / the threshold_tokens property) instead + # of here. The pre-floor value is kept so update_model() can + # re-derive for a new window (switching small -> large must drop + # back to the configured value). + self._config_context_length = config_context_length self._configured_threshold_percent = self.threshold_percent - self.threshold_percent = self._effective_threshold_percent( - self.context_length, self._base_threshold_percent, - ) - threshold_percent = self.threshold_percent - # Floor: never compress below MINIMUM_CONTEXT_LENGTH tokens even if - # the percentage would suggest a lower value. This prevents premature - # compression on large-context models at 50% while keeping the % sane - # for models right at the minimum. _compute_threshold_tokens also - # guards the degenerate case where the floor would equal/exceed the - # window (small models), so auto-compression can still fire (#14690). - self.threshold_tokens = self._compute_threshold_tokens( - self.context_length, threshold_percent, self.max_tokens, - ) - # Apply absolute token cap (compression.threshold_tokens) — takes - # the lower of the ratio-based threshold and the cap. - self._apply_threshold_tokens_cap() + self._resolved_context_length: int | None = None + self._threshold_tokens: int | None = None + self._tail_token_budget: int | None = None + self._max_summary_tokens: int | None = None self.compression_count = 0 - # Derive token budgets: ratio is relative to the threshold, not total context - target_tokens = int(self.threshold_tokens * self.summary_target_ratio) - self.tail_token_budget = target_tokens - self.max_summary_tokens = min( - int(self.context_length * 0.05), _SUMMARY_TOKENS_CEILING, - ) - if not quiet_mode: logger.info( "Context compressor initialized: model=%s context_length=%d " diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index 6364441f2a24..bab0778ae1aa 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -34,6 +34,9 @@ def compressor(): protect_last_n=2, quiet_mode=True, ) + # Resolve context_length while the mock is still active so the + # fixture returns a fully-initialized compressor. + _ = c.context_length return c @@ -2584,11 +2587,14 @@ class TestSummaryTargetRatio: """Tail token budget should be threshold_tokens * summary_target_ratio.""" with patch("agent.context_compressor.get_model_context_length", return_value=200_000): c = ContextCompressor(model="test", quiet_mode=True, summary_target_ratio=0.40) + # Resolve while mock is active (lazy init defers this past __init__). + _ = c.context_length # 200K < 512K → threshold floored at 75%: 150K * 0.40 ratio = 60K assert c.tail_token_budget == 60_000 with patch("agent.context_compressor.get_model_context_length", return_value=1_000_000): c = ContextCompressor(model="test", quiet_mode=True, summary_target_ratio=0.40) + _ = c.context_length # 1M * 0.50 threshold * 0.40 ratio = 200K assert c.tail_token_budget == 200_000 @@ -2596,10 +2602,12 @@ class TestSummaryTargetRatio: """Max summary tokens should be 5% of context, capped at 10K.""" with patch("agent.context_compressor.get_model_context_length", return_value=200_000): c = ContextCompressor(model="test", quiet_mode=True) + _ = c.context_length assert c.max_summary_tokens == 10_000 # 200K * 0.05 with patch("agent.context_compressor.get_model_context_length", return_value=1_000_000): c = ContextCompressor(model="test", quiet_mode=True) + _ = c.context_length assert c.max_summary_tokens == 10_000 # capped at 10K ceiling def test_ratio_clamped(self): @@ -2616,6 +2624,7 @@ class TestSummaryTargetRatio: """Sub-512K models get the 75% small-context threshold floor.""" with patch("agent.context_compressor.get_model_context_length", return_value=100_000): c = ContextCompressor(model="test", quiet_mode=True) + _ = c.context_length assert c.threshold_percent == 0.75 # 75% of 100K = 75K, above the 64K minimum floor assert c.threshold_tokens == 75_000 @@ -2624,6 +2633,7 @@ class TestSummaryTargetRatio: """At 512K+ the configured (default 50%) percentage is used directly.""" with patch("agent.context_compressor.get_model_context_length", return_value=512_000): c = ContextCompressor(model="test", quiet_mode=True) + _ = c.context_length assert c.threshold_percent == 0.50 assert c.threshold_tokens == 256_000 @@ -3492,6 +3502,61 @@ class TestTruncateToolCallArgsJson: assert parsed["content"].endswith("...[truncated]") +class TestLazyContextResolution: + """Verify that ContextCompressor defers get_model_context_length until + context_length is first accessed, so construction never blocks on network + I/O or blocks startup when the model metadata service is slow.""" + + def test_init_does_not_call_get_model_context_length(self): + """get_model_context_length must NOT be called during __init__; it + should only be called on first access of .context_length.""" + with patch( + "agent.context_compressor.get_model_context_length", + ) as mock_get: + c = ContextCompressor( + model="test/model", + threshold_percent=0.85, + protect_first_n=2, + protect_last_n=2, + quiet_mode=True, + ) + mock_get.assert_not_called() + + # First access triggers resolution + _ = c.context_length + mock_get.assert_called_once() + + def test_context_length_setter_bypasses_resolution(self): + """Assigning to .context_length directly must skip the network probe + entirely and return the assigned value.""" + with patch( + "agent.context_compressor.get_model_context_length", + ) as mock_get: + c = ContextCompressor( + model="test/model", + quiet_mode=True, + ) + c.context_length = 100_000 + result = c.context_length + mock_get.assert_not_called() + assert result == 100_000 + + def test_config_context_length_skips_network_probe(self): + """When config_context_length is provided, the resolver must use it + as the cached value and not make a network call.""" + with patch( + "agent.context_compressor.get_model_context_length", + side_effect=lambda model, **kwargs: kwargs.get("config_context_length"), + ) as mock_get: + c = ContextCompressor( + model="test/model", + quiet_mode=True, + config_context_length=200_000, + ) + result = c.context_length + assert result == 200_000 + + class TestPreflightSentinelGuard: """Regression for #36718: the preflight token-display seed in run_conversation must NOT overwrite the -1 sentinel that