mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-22 16:25:58 +00:00
fix(mcp): allow background discovery retry after a run that connected nothing
start_background_mcp_discovery() sets _mcp_discovery_started once and never resets it. If the first background run exits without connecting any MCP server (startup cancellation, OOM restart, transient network failure), every later call returns immediately and the process is permanently stuck with zero MCP tools until a full restart. Fix: when discovery is marked started but the thread is dead and no server is connected, reset the flag and spawn a fresh discovery thread. Also log a WARNING when a discovery run completes with zero connected servers, so the condition is visible instead of silent. Caught in production on a long-running gateway fleet where a gateway restarted under memory pressure and came back with all MCP tools missing.
This commit is contained in:
parent
fbe086f7cc
commit
4fda7efcbd
2 changed files with 111 additions and 2 deletions
|
|
@ -25,12 +25,35 @@ def _has_configured_mcp_servers() -> bool:
|
|||
|
||||
|
||||
def start_background_mcp_discovery(*, logger, thread_name: str) -> None:
|
||||
"""Spawn one shared background MCP discovery thread for this process."""
|
||||
"""Spawn one shared background MCP discovery thread for this process.
|
||||
|
||||
If the first background discovery run exits without connecting any MCP
|
||||
server (for example after startup cancellation / OOM restart), later calls
|
||||
are allowed to retry instead of permanently pinning the process in a
|
||||
"discovery already started" state with zero MCP tools.
|
||||
"""
|
||||
global _mcp_discovery_started, _mcp_discovery_thread
|
||||
|
||||
with _mcp_discovery_lock:
|
||||
if _mcp_discovery_started:
|
||||
return
|
||||
thread = _mcp_discovery_thread
|
||||
if thread is not None and thread.is_alive():
|
||||
return
|
||||
try:
|
||||
from tools.mcp_tool import get_mcp_status
|
||||
|
||||
status = get_mcp_status() or []
|
||||
if any(entry.get("connected") for entry in status):
|
||||
return
|
||||
except Exception:
|
||||
return
|
||||
logger.warning(
|
||||
"Background MCP discovery previously exited with no connected "
|
||||
"servers; retrying discovery thread"
|
||||
)
|
||||
_mcp_discovery_started = False
|
||||
_mcp_discovery_thread = None
|
||||
|
||||
_mcp_discovery_started = True
|
||||
if not _has_configured_mcp_servers():
|
||||
return
|
||||
|
|
@ -38,8 +61,21 @@ def start_background_mcp_discovery(*, logger, thread_name: str) -> None:
|
|||
def _discover() -> None:
|
||||
try:
|
||||
_discover_mcp_tools_without_interactive_oauth()
|
||||
try:
|
||||
from tools.mcp_tool import get_mcp_status
|
||||
status = get_mcp_status() or []
|
||||
if not any(entry.get("connected") for entry in status):
|
||||
logger.warning(
|
||||
"Background MCP discovery completed with zero connected servers"
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("Failed to inspect MCP status after background discovery", exc_info=True)
|
||||
except Exception:
|
||||
logger.debug("Background MCP tool discovery failed", exc_info=True)
|
||||
finally:
|
||||
with _mcp_discovery_lock:
|
||||
global _mcp_discovery_thread, _mcp_discovery_started
|
||||
_mcp_discovery_thread = None
|
||||
|
||||
thread = threading.Thread(
|
||||
target=_discover,
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue