mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-23 16:36:23 +00:00
fix(context): revalidate Codex OAuth context windows
This commit is contained in:
parent
279be8211d
commit
8a0701ca48
2 changed files with 102 additions and 92 deletions
|
|
@ -581,8 +581,13 @@ def _skip_persistent_context_cache(base_url: str, provider: str) -> bool:
|
|||
|
||||
LM Studio excludes caching because loaded context is transient — the user
|
||||
can reload the model with a different context_length at any time.
|
||||
"""
|
||||
return provider == "lmstudio"
|
||||
|
||||
Codex OAuth excludes caching because its context window is account- and
|
||||
entitlement-specific metadata supplied by the authenticated /models
|
||||
endpoint. A fallback value written after a transient probe failure must
|
||||
not prevent a later live probe from observing an updated allocation.
|
||||
"""
|
||||
return (provider or "").strip().lower() in {"lmstudio", "openai-codex"}
|
||||
|
||||
|
||||
def _maybe_cache_local_context_length(
|
||||
|
|
@ -1974,27 +1979,31 @@ def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
|
|||
return result
|
||||
|
||||
|
||||
def _resolve_codex_oauth_context_length(
|
||||
def _resolve_codex_oauth_context_length_with_source(
|
||||
model: str, access_token: str = ""
|
||||
) -> Optional[int]:
|
||||
) -> Tuple[Optional[int], str]:
|
||||
"""Resolve a Codex OAuth model's real context window.
|
||||
|
||||
Prefers a live probe of chatgpt.com/backend-api/codex/models (when we
|
||||
have a bearer token), then falls back to ``_CODEX_OAUTH_CONTEXT_FALLBACK``.
|
||||
|
||||
Returns ``(context_length, source)`` where source is ``"live"`` for a
|
||||
value returned by the authenticated endpoint or ``"fallback"`` for the
|
||||
static conservative table. Callers must not persist the latter.
|
||||
"""
|
||||
model_bare = _strip_provider_prefix(model).strip()
|
||||
if not model_bare:
|
||||
return None
|
||||
return None, ""
|
||||
|
||||
if access_token:
|
||||
live = _fetch_codex_oauth_context_lengths(access_token)
|
||||
if model_bare in live:
|
||||
return live[model_bare]
|
||||
return live[model_bare], "live"
|
||||
# Case-insensitive match in case casing drifts
|
||||
model_lower = model_bare.lower()
|
||||
for slug, ctx in live.items():
|
||||
if slug.lower() == model_lower:
|
||||
return ctx
|
||||
return ctx, "live"
|
||||
|
||||
# Fallback: longest-key-first substring match over hardcoded defaults.
|
||||
model_lower = model_bare.lower()
|
||||
|
|
@ -2002,9 +2011,19 @@ def _resolve_codex_oauth_context_length(
|
|||
_CODEX_OAUTH_CONTEXT_FALLBACK.items(), key=lambda x: len(x[0]), reverse=True
|
||||
):
|
||||
if slug in model_lower:
|
||||
return ctx
|
||||
return ctx, "fallback"
|
||||
|
||||
return None
|
||||
return None, ""
|
||||
|
||||
|
||||
def _resolve_codex_oauth_context_length(
|
||||
model: str, access_token: str = ""
|
||||
) -> Optional[int]:
|
||||
"""Resolve a Codex OAuth model's context length (compatibility wrapper)."""
|
||||
context_length, _source = _resolve_codex_oauth_context_length_with_source(
|
||||
model, access_token=access_token,
|
||||
)
|
||||
return context_length
|
||||
|
||||
|
||||
def _resolve_nous_context_length(
|
||||
|
|
@ -2094,9 +2113,9 @@ def get_model_context_length(
|
|||
Resolution order:
|
||||
0. Explicit config override (model.context_length or custom_providers per-model)
|
||||
0c. Endpoint-scoped metadata for models validated on one multiplexed endpoint
|
||||
1. Persistent cache (previously discovered via probing). Nous URLs
|
||||
bypass the cache here so step 5b can always reconcile against
|
||||
the authoritative portal /v1/models response.
|
||||
1. Persistent cache (previously discovered via probing). Nous URLs,
|
||||
LM Studio, and Codex OAuth bypass the cache here so their provider
|
||||
metadata can be reconciled against the authoritative live source.
|
||||
1b. AWS Bedrock static table (must precede custom-endpoint probe)
|
||||
2. Active endpoint metadata (/models for explicit custom endpoints)
|
||||
3. Local server query (for local endpoints)
|
||||
|
|
@ -2196,24 +2215,13 @@ def get_model_context_length(
|
|||
# LM Studio is excluded — its loaded context length is transient (the
|
||||
# user can reload the model with a different context_length at any time
|
||||
# via /api/v1/models/load), so a stale cached value would mask reloads.
|
||||
# Codex OAuth is excluded because the authenticated /models catalogue is
|
||||
# account-specific and a fallback must never suppress later revalidation.
|
||||
if base_url and not _skip_persistent_context_cache(base_url, provider):
|
||||
cached = get_cached_context_length(model, base_url)
|
||||
if cached is not None:
|
||||
# Invalidate stale Codex OAuth cache entries: pre-PR #14935 builds
|
||||
# resolved gpt-5.x to the direct-API value (e.g. 1.05M) via
|
||||
# models.dev and persisted it. Codex OAuth caps at 272K for every
|
||||
# slug, so any cached Codex entry at or above 400K is a leftover
|
||||
# from the old resolution path. Drop it and fall through to the
|
||||
# live /models probe in step 5 below.
|
||||
if provider == "openai-codex" and cached >= 400_000:
|
||||
logger.info(
|
||||
"Dropping stale Codex cache entry %s@%s -> %s (pre-fix value); "
|
||||
"re-resolving via live /models probe",
|
||||
model, base_url, f"{cached:,}",
|
||||
)
|
||||
_invalidate_cached_context_length(model, base_url)
|
||||
# Invalidate stale 32k cache entries for Kimi-family models.
|
||||
elif cached <= 32768 and _model_name_suggests_kimi(model):
|
||||
if cached <= 32768 and _model_name_suggests_kimi(model):
|
||||
logger.info(
|
||||
"Dropping stale Kimi cache entry %s@%s -> %s (OpenRouter underreport); "
|
||||
"re-resolving via hardcoded defaults",
|
||||
|
|
@ -2452,9 +2460,14 @@ def get_model_context_length(
|
|||
# Codex OAuth enforces lower context limits than the direct OpenAI
|
||||
# API for the same slug (e.g. gpt-5.5 is 1.05M on the API but 272K
|
||||
# on Codex). Authoritative source is Codex's own /models endpoint.
|
||||
codex_ctx = _resolve_codex_oauth_context_length(model, access_token=api_key or "")
|
||||
codex_ctx, codex_source = _resolve_codex_oauth_context_length_with_source(
|
||||
model, access_token=api_key or "",
|
||||
)
|
||||
if codex_ctx:
|
||||
if base_url:
|
||||
# Only a successful authenticated catalogue response is safe to
|
||||
# persist. The static fallback is deliberately runtime-only so a
|
||||
# transient OAuth/network failure cannot poison future probes.
|
||||
if base_url and codex_source == "live":
|
||||
save_context_length(model, base_url, codex_ctx)
|
||||
return codex_ctx
|
||||
if effective_provider == "gmi" and base_url:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue