mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
fix(agent): size the Ollama context window from /api/ps, not the trained max
Ollama sizes a model's real context window by free VRAM at load time, often far below the GGUF trained max that /api/show reports, and its OpenAI-compatible endpoint has no options passthrough — per-request num_ctx and keep_alive are silently dropped, so the window cannot be controlled from the client. The compressor was being sized to the trained max (e.g. 262K for a model actually running at 32K). - Add query_ollama_loaded_context() reading the effective window from /api/ps (60s cache, never persisted — transient load state). - Reconcile after each successful response via sync_ollama_loaded_context(): resize the compressor to the loaded window and warn when it is below the tool-use minimum. Selection surfaces keep showing the trained max; explicit model.context_length still wins. No-op for non-Ollama providers. - Refresh model.ollama_keep_alive through the native API (rate-limited /api/generate ping) since /v1 drops it. - Remove the inert num_ctx/keep_alive request-body plumbing; warn that model.ollama_num_ctx has no effect and point at OLLAMA_CONTEXT_LENGTH / Modelfile num_ctx. Keep detection for the pre-flight window check. - Disable thinking on Ollama thinking models via reasoning_effort 'none' — the only switch its /v1 handler parses (think is dropped).
This commit is contained in:
parent
0e4598b271
commit
1908dd09fc
13 changed files with 540 additions and 96 deletions
|
|
@ -143,12 +143,12 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str
|
|||
f"Ollama loaded `{model}` with only {runtime_ctx:,} tokens of runtime "
|
||||
f"context, but Hermes needs at least {MINIMUM_CONTEXT_LENGTH:,} tokens "
|
||||
"for reliable tool use.\n\n"
|
||||
"Increase the Ollama context for this model and restart/reload the "
|
||||
"model before trying again. A known-good starting point is 65,536 "
|
||||
"tokens. In Hermes config, set `model.ollama_num_ctx: 65536` "
|
||||
"(and `model.context_length: 65536` if you also override the displayed "
|
||||
"model context). If you manage the model through an Ollama Modelfile, "
|
||||
"set `PARAMETER num_ctx 65536` there instead."
|
||||
"Increase the Ollama context for this model and reload it before "
|
||||
"trying again. A known-good starting point is 65,536 tokens: set "
|
||||
"`OLLAMA_CONTEXT_LENGTH=65536` in the Ollama server's environment "
|
||||
"and restart it, or set `PARAMETER num_ctx 65536` in the model's "
|
||||
"Modelfile. (Ollama's OpenAI-compatible API ignores per-request "
|
||||
"num_ctx, so this cannot be fixed from the client side.)"
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -2118,6 +2118,15 @@ def run_conversation(
|
|||
"cache_write_tokens": canonical_usage.cache_write_tokens,
|
||||
"reasoning_tokens": canonical_usage.reasoning_tokens,
|
||||
}
|
||||
# Reconcile with the window Ollama actually loaded before
|
||||
# the usage update: the /v1 endpoint cannot pin num_ctx,
|
||||
# so the served window (VRAM-tiered) is discoverable only
|
||||
# now that the model is resident. No-op for other providers.
|
||||
try:
|
||||
from agent.agent_runtime_helpers import sync_ollama_loaded_context
|
||||
sync_ollama_loaded_context(agent)
|
||||
except Exception as _ollama_sync_exc: # pragma: no cover - defensive
|
||||
logger.debug("Ollama loaded-context sync failed: %s", _ollama_sync_exc)
|
||||
agent.context_compressor.update_from_response(usage_dict)
|
||||
elif getattr(
|
||||
agent.context_compressor,
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue