diff --git a/agent/agent_init.py b/agent/agent_init.py index 495260d2188..ce32645df78 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -2012,6 +2012,22 @@ def init_agent( agent._ollama_num_ctx, ) + # ── Ollama keep_alive ── + # How long the server keeps the model resident after a request (Ollama + # default: 5 minutes). Set model.ollama_keep_alive in config.yaml to a Go + # duration string ("30m", "2h") or seconds; -1 keeps it loaded until the + # server exits. Sent per-request alongside num_ctx. + agent._ollama_keep_alive = None + if isinstance(_model_cfg, dict): + _keep_alive_raw = _model_cfg.get("ollama_keep_alive") + if _keep_alive_raw is not None: + if isinstance(_keep_alive_raw, (int, float)): + agent._ollama_keep_alive = int(_keep_alive_raw) + else: + _keep_alive_str = str(_keep_alive_raw).strip() + if _keep_alive_str: + agent._ollama_keep_alive = _keep_alive_str + # Codex gpt-5.x autoraise notice: show at most once per profile/config # state. Without the persisted marker the notice re-fires on every agent # init — and the gateway rebuilds the agent per inbound message, so Discord diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 9be0a7fed52..1c1deaa88fc 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -879,6 +879,12 @@ def build_api_kwargs(agent, api_messages: list) -> dict: session_id=getattr(agent, "session_id", None), provider_profile=_profile, ollama_num_ctx=agent._ollama_num_ctx, + ollama_keep_alive=getattr(agent, "_ollama_keep_alive", None), + ollama_supports_thinking=( + agent._ollama_supports_thinking_cached() + if (agent.provider or "").strip().lower() == "ollama" + else None + ), # Context forwarded to profile hooks: provider_preferences=_prefs or None, openrouter_min_coding_score=agent.openrouter_min_coding_score, diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 88fab090251..cca79be75a6 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1435,6 +1435,51 @@ def query_ollama_supports_vision(model: str, base_url: str, api_key: str = "") - return None +def query_ollama_supports_thinking(model: str, base_url: str, api_key: str = "") -> Optional[bool]: + """Return True/False when Ollama ``/api/show`` reports thinking support. + + Uses the ``capabilities`` field (Ollama 0.6.0+). Returns None when the + server is unreachable, not Ollama, the model is unknown, or the server + predates the capabilities field — callers should treat None as "unknown" + and not gate on it. + """ + import httpx + + bare_model = _strip_provider_prefix(model) + if not bare_model or not base_url: + return None + + try: + if detect_local_server_type(base_url, api_key=api_key) != "ollama": + return None + except Exception: + return None + + server_url = base_url.rstrip("/") + if server_url.endswith("/v1"): + server_url = server_url[:-3] + + headers = _auth_headers(api_key) + + try: + with httpx.Client(timeout=3.0, headers=headers) as client: + resp = client.post(f"{server_url}/api/show", json={"name": bare_model}) + if resp.status_code != 200: + return None + data = resp.json() + except Exception: + return None + + caps = data.get("capabilities") + if isinstance(caps, list): + if any(str(cap).lower() == "thinking" for cap in caps): + return True + if caps: + return False + + return None + + def _query_ollama_api_show(model: str, base_url: str, api_key: str = "") -> Optional[int]: """Query an Ollama server's native ``/api/show`` for context length. diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index ff2cdcbaee6..a7c0207f9bc 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -256,6 +256,8 @@ class ChatCompletionsTransport(ProviderTransport): is_lmstudio: bool is_custom_provider: bool ollama_num_ctx: int | None + ollama_keep_alive: int | str | None + ollama_supports_thinking: bool | None # Provider routing provider_preferences: dict | None # Qwen-specific @@ -534,6 +536,8 @@ class ChatCompletionsTransport(ProviderTransport): model=model, base_url=params.get("base_url"), ollama_num_ctx=params.get("ollama_num_ctx"), + ollama_keep_alive=params.get("ollama_keep_alive"), + ollama_supports_thinking=params.get("ollama_supports_thinking"), session_id=params.get("session_id"), ) ) diff --git a/apps/desktop/src/app/chat/composer/status-stack/index.tsx b/apps/desktop/src/app/chat/composer/status-stack/index.tsx index fd360764da4..f3ea58dd77e 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/index.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/index.tsx @@ -23,6 +23,7 @@ import { $previewStatusBySession, dismissPreviewArtifact } from '@/store/preview import { $threadScrolledUp } from '@/store/thread-scroll' import { openSessionInNewWindow } from '@/store/windows' +import { OllamaColdStartRow, useOllamaColdStart } from './ollama-cold-start' import { PreviewStatusRow } from './preview-row' import { StatusItemRow } from './status-row' @@ -173,6 +174,13 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro sections.push({ key: 'preview', node: previewBlock }) } + // Cold-start feedback while a local Ollama model loads its weights. + const ollamaColdStartModel = useOllamaColdStart() + + if (ollamaColdStartModel) { + sections.push({ key: 'ollama-cold-start', node: }) + } + if (queue) { sections.push({ key: 'queue', node: queue }) } diff --git a/apps/desktop/src/app/chat/composer/status-stack/ollama-cold-start.tsx b/apps/desktop/src/app/chat/composer/status-stack/ollama-cold-start.tsx new file mode 100644 index 00000000000..58bf86ffadb --- /dev/null +++ b/apps/desktop/src/app/chat/composer/status-stack/ollama-cold-start.tsx @@ -0,0 +1,91 @@ +import { useStore } from '@nanostores/react' +import { useEffect, useState } from 'react' + +import { GlyphSpinner } from '@/components/ui/glyph-spinner' +import { getOllamaModels } from '@/hermes' +import { useI18n } from '@/i18n' +import { $awaitingResponse, $currentModel, $currentProvider } from '@/store/session' + +// Wait this long after a turn starts before concluding the silence is a cold +// model load (a warm model answers well inside it), then confirm against the +// server before showing anything. +const COLD_START_GRACE_MS = 4_000 +const POLL_INTERVAL_MS = 2_000 + +/** + * Cold-start detection for local Ollama models: returns the model name while + * a turn has been awaiting its first token past the grace period and the + * target model is not yet in the server's loaded list — i.e. Ollama is + * reading weights into memory. Returns '' for non-Ollama providers and for + * warm models (the loaded check keeps ordinary slow generations quiet). + * + * A hook rather than a self-hiding component so the status stack can count + * it toward its own visibility (the stack collapses when every section is + * empty). + */ +export function useOllamaColdStart(): string { + const awaiting = useStore($awaitingResponse) + const provider = useStore($currentProvider) + const model = useStore($currentModel) + const [loadingModel, setLoadingModel] = useState('') + + const active = awaiting && provider === 'ollama' && Boolean(model) + + useEffect(() => { + if (!active) { + setLoadingModel('') + + return + } + + let cancelled = false + let timer: ReturnType + + const check = async () => { + try { + const data = await getOllamaModels() + + if (cancelled) { + return + } + + const loaded = data.running.some(r => r.name === model || r.name.split(':', 1)[0] === model.split(':', 1)[0]) + + // Loaded (or server unreachable — nothing useful to say): clear and + // let the normal streaming UI take over. + if (loaded || !data.reachable) { + setLoadingModel('') + + return + } + + setLoadingModel(model) + timer = setTimeout(() => void check(), POLL_INTERVAL_MS) + } catch { + if (!cancelled) { + setLoadingModel('') + } + } + } + + timer = setTimeout(() => void check(), COLD_START_GRACE_MS) + + return () => { + cancelled = true + clearTimeout(timer) + } + }, [active, model]) + + return active ? loadingModel : '' +} + +export function OllamaColdStartRow({ model }: { model: string }) { + const { t } = useI18n() + + return ( +
+ + {t.statusStack.ollamaLoading(model)} +
+ ) +} diff --git a/apps/desktop/src/app/settings/ollama-panel.tsx b/apps/desktop/src/app/settings/ollama-panel.tsx new file mode 100644 index 00000000000..36befbaec9b --- /dev/null +++ b/apps/desktop/src/app/settings/ollama-panel.tsx @@ -0,0 +1,364 @@ +import { useCallback, useEffect, useRef, useState } from 'react' + +import { Button } from '@/components/ui/button' +import { Input } from '@/components/ui/input' +import { RowButton } from '@/components/ui/row-button' +import { + deleteOllamaModel, + getOllamaModels, + loadOllamaModel, + pollOllamaPull, + startOllamaPull +} from '@/hermes' +import { useI18n } from '@/i18n' +import { Check, ChevronDown, Cpu, Download, Loader2, Trash2 } from '@/lib/icons' +import { cn } from '@/lib/utils' +import { notify, notifyError } from '@/store/notifications' +import type { OllamaModelsResponse, OllamaPullStatus } from '@/types/hermes' + +import { CONTROL_TEXT } from './constants' +import { SettingsCategoryHeading } from './env-credentials' +import { Pill } from './primitives' + +const PULL_POLL_INTERVAL_MS = 750 + +function formatBytes(bytes?: number): string { + if (!bytes || bytes <= 0) { + return '' + } + + const gib = bytes / 1024 ** 3 + + return gib >= 1 ? `${gib.toFixed(1)} GB` : `${Math.round(bytes / 1024 ** 2)} MB` +} + +function formatContext(tokens?: number): string { + if (!tokens || tokens <= 0) { + return '' + } + + return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : String(tokens) +} + +/** + * Local Ollama server card for the Providers page. Same row language as the + * OAuth provider cards, but the connection kind is reachability rather than + * a credential: the status tag shows "Running · N models" or a start-the- + * server hint. Expanding a running server reveals model management — + * installed models (size, delete, warm-up), loaded models (VRAM, context), + * curated pull recommendations, and free-form pull. + */ +export function OllamaProviderCard() { + const { t } = useI18n() + const copy = t.settings.ollama + const [data, setData] = useState(null) + const [expanded, setExpanded] = useState(false) + const [pull, setPull] = useState(null) + const [pullModel, setPullModel] = useState('') + const [busyModel, setBusyModel] = useState(null) + const pollTimer = useRef>(null) + + const refresh = useCallback(async () => { + try { + setData(await getOllamaModels()) + } catch { + setData(null) + } + }, []) + + useEffect(() => { + void refresh() + + return () => { + if (pollTimer.current) { + clearTimeout(pollTimer.current) + } + } + }, [refresh]) + + // While the server is down (or detection failed), poll so the card flips + // to "Running" by itself when the user starts Ollama — the status endpoint + // answers in ~20ms, so this is cheap. Stops as soon as it's reachable. + const reachableNow = Boolean(data?.reachable) + + useEffect(() => { + if (reachableNow) { + return + } + + const timer = setInterval(() => void refresh(), 5_000) + + return () => clearInterval(timer) + }, [reachableNow, refresh]) + + const pollUntilDone = useCallback( + (jobId: string) => { + const tick = async () => { + let status: OllamaPullStatus + + try { + status = await pollOllamaPull(jobId) + } catch (err) { + setPull(null) + notifyError(err, copy.pullFailed) + + return + } + + setPull(status) + + if (status.status === 'pulling') { + pollTimer.current = setTimeout(() => void tick(), PULL_POLL_INTERVAL_MS) + + return + } + + if (status.status === 'done') { + notify({ kind: 'success', title: copy.pullDone(status.model), message: '' }) + setPull(null) + setPullModel('') + void refresh() + } else { + notifyError(new Error(status.error_message || status.detail || 'pull failed'), copy.pullFailed) + setPull(null) + } + } + + void tick() + }, + [copy, refresh] + ) + + const beginPull = useCallback( + async (model: string) => { + const name = model.trim() + + if (!name || pull) { + return + } + + try { + const { job_id } = await startOllamaPull(name) + + setPull({ job_id, model: name, status: 'pulling', detail: 'starting' }) + pollUntilDone(job_id) + } catch (err) { + notifyError(err, copy.pullFailed) + } + }, + [copy.pullFailed, pollUntilDone, pull] + ) + + const removeModel = useCallback( + async (model: string) => { + setBusyModel(model) + + try { + const result = await deleteOllamaModel(model) + + if (!result.ok) { + notifyError(new Error(result.message), copy.deleteFailed) + } else { + void refresh() + } + } catch (err) { + notifyError(err, copy.deleteFailed) + } finally { + setBusyModel(null) + } + }, + [copy.deleteFailed, refresh] + ) + + const warmModel = useCallback( + async (model: string) => { + setBusyModel(model) + + try { + const result = await loadOllamaModel(model) + + if (!result.ok) { + notifyError(new Error(result.message), copy.loadFailed) + } else { + void refresh() + } + } catch (err) { + notifyError(err, copy.loadFailed) + } finally { + setBusyModel(null) + } + }, + [copy.loadFailed, refresh] + ) + + const reachable = Boolean(data?.reachable) + const installedCount = data?.installed.length ?? 0 + const running = new Map((data?.running ?? []).map(r => [r.name, r])) + + const pullPercent = + pull?.total_bytes && pull.completed_bytes !== undefined + ? Math.min(100, Math.round((pull.completed_bytes / pull.total_bytes) * 100)) + : null + + // Detection still pending: render nothing rather than flashing the + // not-running hint at every page open on machines without Ollama. + if (data === null) { + return null + } + + return ( +
+ +
+ setExpanded(open => !open)} + > +
+
+ Ollama + {reachable ? ( + + + {copy.running(installedCount)} + + ) : null} +
+

+ {reachable ? copy.desc : copy.notRunning} +

+
+ {reachable && ( + + )} +
+ + {reachable && expanded && data && ( +
+ {data.kv_cache_advisory && ( +

+ {copy.kvCacheAdvisory( + data.kv_cache_advisory.model, + formatContext(data.kv_cache_advisory.loaded_context), + formatContext(data.kv_cache_advisory.trained_context) + )} +

+ )} + +
+ {data.installed.map(model => { + const live = running.get(model.name) + const busy = busyModel === model.name + + const meta = [model.parameter_size, model.quantization, formatBytes(model.size_bytes)] + .filter(Boolean) + .join(' · ') + + return ( +
+
+
+ {model.name} + {live && ( + + {copy.loaded} + {live.context_length ? ` · ${formatContext(live.context_length)}` : ''} + {live.size_vram_bytes ? ` · ${formatBytes(live.size_vram_bytes)} VRAM` : ''} + + )} +
+ {meta &&

{meta}

} +
+
+ {!live && ( + + )} + +
+
+ ) + })} +
+ +
+

{copy.getModels}

+ {pull ? ( +
+
+ + {pull.model} + + {pullPercent !== null ? `${pullPercent}%` : pull.detail || ''} + +
+ {pullPercent !== null && ( +
+
+
+ )} +
+ ) : ( + <> +
+ {data.recommended.map(rec => ( +
+
+ {rec.model} +

{rec.description}

+
+ +
+ ))} +
+
+ setPullModel(event.target.value)} + onKeyDown={event => { + if (event.key === 'Enter') { + void beginPull(pullModel) + } + }} + placeholder={copy.pullPlaceholder} + value={pullModel} + /> + +
+ + )} +
+
+ )} +
+
+ ) +} diff --git a/apps/desktop/src/app/settings/providers-settings.tsx b/apps/desktop/src/app/settings/providers-settings.tsx index 214cf37a960..a361922b5fc 100644 --- a/apps/desktop/src/app/settings/providers-settings.tsx +++ b/apps/desktop/src/app/settings/providers-settings.tsx @@ -26,6 +26,7 @@ import type { EnvVarInfo, OAuthProvider } from '@/types/hermes' import { isKeyVar, ProviderKeyRows } from './credential-key-ui' import { SettingsCategoryHeading, useEnvCredentials } from './env-credentials' import { providerGroup, providerMeta, providerPriority } from './helpers' +import { OllamaProviderCard } from './ollama-panel' import { LoadingState, SettingsContent } from './primitives' // The embedded terminal (and thus the "run disconnect command" path) only @@ -457,6 +458,10 @@ export function ProvidersSettings({ onClose, onViewChange, view }: ProvidersSett onWantApiKey={() => onViewChange('keys')} providers={oauthProviders} /> + {/* Local Ollama server — a provider whose connection kind is + reachability rather than a credential, so it renders its own card + instead of an OAuth or API-key row. */} + ) } diff --git a/apps/desktop/src/components/model-picker.tsx b/apps/desktop/src/components/model-picker.tsx index 37de510c653..b06fab64de6 100644 --- a/apps/desktop/src/components/model-picker.tsx +++ b/apps/desktop/src/components/model-picker.tsx @@ -1,11 +1,11 @@ import { useQuery } from '@tanstack/react-query' -import { useState } from 'react' +import { useEffect, useState } from 'react' import { useI18n } from '@/i18n' import { requestModelOptions } from '@/lib/model-options' import { currentPickerSelection } from '@/lib/model-status-label' import { normalize } from '@/lib/text' -import type { ModelOptionProvider, ModelPricing } from '@/types/hermes' +import type { ModelCapabilities, ModelOptionProvider, ModelPricing } from '@/types/hermes' import type { HermesGateway } from '../hermes' import { cn } from '../lib/utils' @@ -61,6 +61,30 @@ export function ModelPickerDialog({ const providers = modelOptions.data?.providers ?? [] + // Local Ollama capability metadata is backfilled server-side off the + // request path (the payload returns immediately; native /api/show data + // lands moments later). When an ollama row is missing its enrichment + // (no tools flag), refetch once shortly after so badges appear in-place + // instead of on the next open. + const ollamaUnenriched = providers.some( + p => + p.slug === 'ollama' && + (p.models?.length ?? 0) > 0 && + p.models?.some(m => p.capabilities?.[m]?.tools === undefined) + ) + + const refetchOptions = modelOptions.refetch + + useEffect(() => { + if (!open || !ollamaUnenriched) { + return + } + + const timer = setTimeout(() => void refetchOptions(), 1_500) + + return () => clearTimeout(timer) + }, [open, ollamaUnenriched, refetchOptions]) + const { model: optionsModel, provider: optionsProvider } = currentPickerSelection( !!sessionId, { model: currentModel, provider: currentProvider }, @@ -205,6 +229,10 @@ function ModelResults({ const isCurrent = model === currentModel && provider.slug === currentProvider const price = provider.pricing?.[model] const locked = unavailable.has(model) + const caps = provider.capabilities?.[model] + // Only an explicit tools:false demotes — absent means unknown, + // and models.dev gaps must not smear working models. + const noTools = caps?.tools === false return ( {model} + {noTools && ( + + {copy.noTools} + + )} + {locked && ( {copy.pro} )} @@ -243,6 +286,44 @@ function ModelResults({ ) } +// Compact local-model metadata: context window (agent-critical) plus +// parameter size / quantization when the native server reported them. +// Renders nothing for models with no metadata beyond the fast/reasoning flags. +function ModelMeta({ caps, isCurrent }: { caps?: ModelCapabilities; isCurrent: boolean }) { + if (!caps) { + return null + } + + const parts: string[] = [] + + if (caps.context_length) { + parts.push(caps.context_length >= 1024 ? `${Math.round(caps.context_length / 1024)}k` : String(caps.context_length)) + } + + if (caps.parameter_size) { + parts.push(caps.parameter_size) + } + + if (caps.quantization) { + parts.push(caps.quantization) + } + + if (parts.length === 0) { + return null + } + + return ( + + {parts.join(' · ')} + + ) +} + // Compact In/Out $/Mtok price tag, mirroring the CLI picker's price columns. // Renders nothing when pricing is unavailable for the model. function ModelPrice({ price, isCurrent }: { price?: ModelPricing; isCurrent: boolean }) { diff --git a/apps/desktop/src/components/onboarding/index.test.tsx b/apps/desktop/src/components/onboarding/index.test.tsx index 35cac9c4d17..302c0f2c224 100644 --- a/apps/desktop/src/components/onboarding/index.test.tsx +++ b/apps/desktop/src/components/onboarding/index.test.tsx @@ -1,4 +1,6 @@ -import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { QueryClient, QueryClientProvider } from '@tanstack/react-query' +import { cleanup, fireEvent, render as rtlRender, screen } from '@testing-library/react' +import type { ReactElement } from 'react' import { afterEach, describe, expect, it } from 'vitest' import { $desktopOnboarding, type DesktopOnboardingState, type OnboardingContext } from '@/store/onboarding' @@ -6,6 +8,15 @@ import type { OAuthProvider } from '@/types/hermes' import { Picker } from '.' +// The Picker's local-server detection uses react-query; queries stay pending +// in tests (no window.hermesDesktop bridge), which renders as "no detected +// row" — the same degraded state as detection failing in the app. +function render(ui: ReactElement) { + const client = new QueryClient({ defaultOptions: { queries: { retry: false } } }) + + return rtlRender({ui}) +} + function provider(id: string, name = id): OAuthProvider { return { cli_command: `hermes login ${id}`, diff --git a/apps/desktop/src/components/onboarding/index.tsx b/apps/desktop/src/components/onboarding/index.tsx index 0b81db3775b..b344448925e 100644 --- a/apps/desktop/src/components/onboarding/index.tsx +++ b/apps/desktop/src/components/onboarding/index.tsx @@ -1,10 +1,11 @@ import { useStore } from '@nanostores/react' +import { useQuery } from '@tanstack/react-query' import { useEffect, useMemo, useRef, useState } from 'react' import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' import { Input } from '@/components/ui/input' -import { getGlobalModelOptions } from '@/hermes' +import { detectLocalServers, getGlobalModelOptions } from '@/hermes' import { useI18n } from '@/i18n' import { Check, ChevronDown, ChevronLeft, KeyRound, Loader2 } from '@/lib/icons' import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors' @@ -22,13 +23,14 @@ import { peekPendingProviderOAuth, refreshOnboarding, saveOnboardingApiKey, + saveOnboardingDetectedOllama, setOnboardingMode, startProviderOAuth } from '@/store/onboarding' import type { ModelOptionProvider, OAuthProvider } from '@/types/hermes' import { DocsLink, FlowPanel, Status } from './flow' -import { FeaturedProviderRow, KeyProviderRow, ProviderRow, sortProviders } from './providers' +import { DetectedLocalServerRow, FeaturedProviderRow, KeyProviderRow, ProviderRow, sortProviders } from './providers' export { FeaturedProviderRow, KeyProviderRow, ProviderRow, providerTitle, sortProviders } from './providers' @@ -399,6 +401,22 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) { const hasOauth = ordered.length > 0 const apiKeyOptions = useApiKeyCatalog() + // Detected local model servers (Ollama). Non-blocking: the picker renders + // immediately and the detected row appears when the probe answers. Failures + // degrade to "no row" — the manual local-endpoint path still exists. + const localServers = useQuery({ + queryKey: ['local-servers-detect'], + queryFn: () => detectLocalServers(), + staleTime: 60_000, + retry: false + }) + + const detectedOllama = useMemo(() => { + const servers = localServers.data?.servers ?? [] + + return servers.find(s => s.type === 'ollama' && s.reachable && s.models.length > 0) ?? null + }, [localServers.data]) + // localEndpoint forces the key form regardless of `mode` (which a manual // provider refresh may flip back to 'oauth'); it preselects the local option // and hides the "back to sign in" link since the user came specifically to @@ -438,6 +456,12 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
{featured ? : null} + {detectedOllama ? ( + saveOnboardingDetectedOllama(baseUrl, model, ctx)} + server={detectedOllama} + /> + ) : null} {showRest ? ( <> {rest.map(p => ( diff --git a/apps/desktop/src/components/onboarding/providers.tsx b/apps/desktop/src/components/onboarding/providers.tsx index f7dafdae95d..3ad9dd5e82d 100644 --- a/apps/desktop/src/components/onboarding/providers.tsx +++ b/apps/desktop/src/components/onboarding/providers.tsx @@ -1,7 +1,10 @@ +import { useState } from 'react' + import { RowButton } from '@/components/ui/row-button' import { useI18n } from '@/i18n' -import { Check, ChevronRight, Terminal } from '@/lib/icons' -import type { OAuthProvider } from '@/types/hermes' +import { Check, ChevronRight, Cpu, Loader2, Terminal } from '@/lib/icons' +import { cn } from '@/lib/utils' +import type { LocalServerInfo, OAuthProvider } from '@/types/hermes' const PROVIDER_DISPLAY: Record = { nous: { order: 0, title: 'Nous Portal' }, @@ -116,3 +119,83 @@ export function ProviderRow({ ) } + +/** + * A local model server found by /api/local-servers/detect (currently the + * Ollama row; LM Studio detection reuses the shape later). Collapsed: a + * provider row showing "detected · N models installed". Expanded: the + * installed-model list, each row a one-click connect — no URL typing, no + * blind first-model assignment. + */ +export function DetectedLocalServerRow({ + onConnect, + server +}: { + onConnect: (baseUrl: string, model: string) => Promise<{ ok: boolean; message?: string }> + server: LocalServerInfo +}) { + const { t } = useI18n() + const copy = t.onboarding.detectedLocal + const [expanded, setExpanded] = useState(false) + const [busyModel, setBusyModel] = useState(null) + const [error, setError] = useState('') + + const connect = async (model: string) => { + if (busyModel) { + return + } + + setBusyModel(model) + setError('') + const result = await onConnect(server.base_url, model) + + if (!result.ok) { + setError(result.message || copy.connectFailed) + setBusyModel(null) + } + // On success the overlay unmounts via completeDesktopOnboarding(). + } + + return ( +
+ setExpanded(open => !open)}> +
+
+ + Ollama + + + {copy.detected} + +
+

{copy.modelsInstalled(server.models.length)}

+
+ +
+ {expanded ? ( +
+ {server.models.map(model => ( + void connect(model)} + > + {model} + {busyModel === model ? ( + + ) : ( + + {copy.use} + + )} + + ))} + {error ?

{error}

: null} +
+ ) : null} +
+ ) +} diff --git a/apps/desktop/src/hermes.ts b/apps/desktop/src/hermes.ts index 3bfd02e276f..b97ba86c8cd 100644 --- a/apps/desktop/src/hermes.ts +++ b/apps/desktop/src/hermes.ts @@ -19,6 +19,7 @@ import type { EnvVarInfo, HermesConfig, HermesConfigRecord, + LocalServersResponse, LogsResponse, McpCatalogResponse, McpServerSummary, @@ -37,6 +38,8 @@ import type { OAuthProvidersResponse, OAuthStartResponse, OAuthSubmitResponse, + OllamaModelsResponse, + OllamaPullStatus, PaginatedSessions, ProfileCreatePayload, ProfileSetupCommand, @@ -902,6 +905,62 @@ export interface RecommendedDefaultModel { free_tier: boolean | null } +// Detect running local model servers (Ollama, LM Studio) by probing their +// well-known ports plus the configured base_url. Results are cached +// server-side for 60s; refresh=true busts the cache. +export function detectLocalServers(opts?: { refresh?: boolean }): Promise { + return window.hermesDesktop.api({ + ...profileScoped(), + path: opts?.refresh ? '/api/local-servers/detect?refresh=1' : '/api/local-servers/detect' + }) +} + +// Installed + running + recommended models for the local Ollama server. +export function getOllamaModels(): Promise { + return window.hermesDesktop.api({ + ...profileScoped(), + path: '/api/ollama/models' + }) +} + +// Start pulling a model from the Ollama registry; poll with pollOllamaPull. +export function startOllamaPull(model: string): Promise<{ job_id: string }> { + return window.hermesDesktop.api<{ job_id: string }>({ + ...profileScoped(), + path: '/api/ollama/pull', + method: 'POST', + body: { model } + }) +} + +export function pollOllamaPull(jobId: string): Promise { + return window.hermesDesktop.api({ + ...profileScoped(), + path: `/api/ollama/pull/${encodeURIComponent(jobId)}` + }) +} + +export function deleteOllamaModel(model: string): Promise<{ ok: boolean; message: string }> { + return window.hermesDesktop.api<{ ok: boolean; message: string }>({ + ...profileScoped(), + path: '/api/ollama/delete', + method: 'POST', + body: { model } + }) +} + +// Load a model into memory. keep_alive pins residence ("5m", "2h", "-1" = +// indefinite); Ollama applies keep_alive per-load, so this doubles as the +// way to change it for an already-loaded model. +export function loadOllamaModel(model: string, keepAlive?: string): Promise<{ ok: boolean; message: string }> { + return window.hermesDesktop.api<{ ok: boolean; message: string }>({ + ...profileScoped(), + path: '/api/ollama/load', + method: 'POST', + body: { model, keep_alive: keepAlive ?? null } + }) +} + // Recommended default model for a freshly-authenticated provider. Mirrors the // curation `hermes model` does — for Nous it honors the free/paid tier so a // free user gets a free model instead of a paid default. diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 0ed3e0ff881..6b1bf4eadcf 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -680,6 +680,24 @@ export const en: Translations = { curator: { label: 'Curator', hint: 'Skill-usage review' } } }, + ollama: { + categoryTitle: 'Local servers', + running: (n: number) => (n === 1 ? 'Running · 1 model' : `Running · ${n} models`), + notRunning: 'Not running — start Ollama (localhost:11434) to manage and use local models.', + desc: 'Pull, delete, and warm up models on the local Ollama server.', + loaded: 'Loaded', + load: 'Load', + loadFailed: 'Could not load the model', + deleteLabel: (model: string) => `Delete ${model}`, + deleteFailed: 'Could not delete the model', + getModels: 'Get models', + pull: 'Pull', + pullPlaceholder: 'model:tag (e.g. qwen3:8b)', + pullDone: (model: string) => `${model} is ready`, + pullFailed: 'Model pull failed', + kvCacheAdvisory: (model: string, loaded: string, trained: string) => + `${model} is running a ${loaded} context but supports ${trained}. Setting OLLAMA_KV_CACHE_TYPE=q8_0 on the Ollama server roughly doubles the context that fits in the same memory.` + }, providers: { connectAccount: 'Connect an account', haveApiKey: 'Have an API key instead?', @@ -1737,6 +1755,7 @@ export const en: Translations = { stop: 'Stop', dismiss: 'Dismiss', exit: code => `exit ${code}`, + ollamaLoading: (model: string) => `Loading ${model} into memory…`, coding: { title: 'Working tree', noBranch: 'No branch', @@ -1896,6 +1915,12 @@ export const en: Translations = { connected: 'Connected', featuredPitch: 'One subscription, 300+ frontier models — the recommended way to run Hermes', openRouterPitch: 'One key, hundreds of models — a solid default', + detectedLocal: { + detected: 'Detected', + modelsInstalled: (n: number) => (n === 1 ? '1 model installed — runs on this machine' : `${n} models installed — runs on this machine`), + use: 'Use', + connectFailed: 'Could not connect to the local server.' + }, apiKeyOptions: { openrouter: { short: 'one key, many models', @@ -1965,6 +1990,8 @@ export const en: Translations = { noAuthenticatedProviders: 'No authenticated providers.', pro: 'Pro', proNeedsSubscription: 'Pro models need a paid Nous subscription.', + noTools: 'No tools', + noToolsTitle: 'This model does not support tool calling — most agent features will not work with it.', free: 'Free', freeTier: 'Free tier', priceTitle: 'Input / Output price per million tokens' diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 10088833cd6..18409777302 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -1874,6 +1874,12 @@ export const ja = defineLocale({ connected: '接続済み', featuredPitch: '1 つのサブスクリプションで 300 以上の最先端モデル — Hermes を実行するための推奨方法', openRouterPitch: '1 つのキーで数百のモデル — 堅実なデフォルト', + detectedLocal: { + detected: '検出済み', + modelsInstalled: (n: number) => `${n} 個のモデルがインストール済み — このマシンで実行`, + use: '使用', + connectFailed: 'ローカルサーバーに接続できませんでした。' + }, apiKeyOptions: { openrouter: { short: '1 つのキーで多くのモデル', @@ -1943,6 +1949,8 @@ export const ja = defineLocale({ noAuthenticatedProviders: '認証済みプロバイダーがありません。', pro: 'Pro', proNeedsSubscription: 'Pro モデルには有料の Nous サブスクリプションが必要です。', + noTools: 'ツール非対応', + noToolsTitle: 'このモデルはツール呼び出しに対応していません — ほとんどのエージェント機能が動作しません。', free: '無料', freeTier: '無料プラン', priceTitle: '100 万トークンあたりの入力/出力価格' diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 5c10c3d6873..bfc9dd2d109 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -580,6 +580,23 @@ export interface Translations { providerDefault: string tasks: Record } + ollama: { + categoryTitle: string + running: (n: number) => string + notRunning: string + desc: string + loaded: string + load: string + loadFailed: string + deleteLabel: (model: string) => string + deleteFailed: string + getModels: string + pull: string + pullPlaceholder: string + pullDone: (model: string) => string + pullFailed: string + kvCacheAdvisory: (model: string, loaded: string, trained: string) => string + } providers: { connectAccount: string haveApiKey: string @@ -1419,6 +1436,7 @@ export interface Translations { stop: string dismiss: string exit: (code: number) => string + ollamaLoading: (model: string) => string coding: { title: string noBranch: string @@ -1554,6 +1572,12 @@ export interface Translations { connected: string featuredPitch: string openRouterPitch: string + detectedLocal: { + detected: string + modelsInstalled: (n: number) => string + use: string + connectFailed: string + } apiKeyOptions: Record backToSignIn: string getKey: string @@ -1605,6 +1629,8 @@ export interface Translations { noAuthenticatedProviders: string pro: string proNeedsSubscription: string + noTools: string + noToolsTitle: string free: string freeTier: string priceTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 095bf6e286f..19bbcf71ac8 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -1817,6 +1817,12 @@ export const zhHant = defineLocale({ connected: '已連線', featuredPitch: '一個訂閱,300+ 前沿模型 — 執行 Hermes 的建議方式', openRouterPitch: '一個金鑰,數百個模型 — 穩定的預設選擇', + detectedLocal: { + detected: '已偵測到', + modelsInstalled: (n: number) => `已安裝 ${n} 個模型 — 在本機執行`, + use: '使用', + connectFailed: '無法連線到本機伺服器。' + }, apiKeyOptions: { openrouter: { short: '一個金鑰,多個模型', description: '用一個金鑰存取數百個模型。適合新安裝的預設選擇。' }, openai: { short: 'GPT 等級模型', description: '直接存取 OpenAI 模型。' }, @@ -1880,6 +1886,8 @@ export const zhHant = defineLocale({ noAuthenticatedProviders: '沒有已驗證的提供方。', pro: 'Pro', proNeedsSubscription: 'Pro 模型需要付費 Nous 訂閱。', + noTools: '不支援工具', + noToolsTitle: '該模型不支援工具呼叫 — 大多數智能體功能將無法使用。', free: '免費', freeTier: '免費層', priceTitle: '每百萬 Token 的輸入/輸出價格' diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 411bfcf6ac1..921a8dc135e 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -868,6 +868,24 @@ export const zh: Translations = { curator: { label: '维护器', hint: '技能使用审查' } } }, + ollama: { + categoryTitle: '本地服务器', + running: (n: number) => `运行中 · ${n} 个模型`, + notRunning: '未运行 — 启动 Ollama(localhost:11434)以管理和使用本地模型。', + desc: '在本地 Ollama 服务器上拉取、删除和预热模型。', + loaded: '已加载', + load: '加载', + loadFailed: '无法加载模型', + deleteLabel: (model: string) => `删除 ${model}`, + deleteFailed: '无法删除模型', + getModels: '获取模型', + pull: '拉取', + pullPlaceholder: 'model:tag(例如 qwen3:8b)', + pullDone: (model: string) => `${model} 已就绪`, + pullFailed: '模型拉取失败', + kvCacheAdvisory: (model: string, loaded: string, trained: string) => + `${model} 当前以 ${loaded} 上下文运行,但支持 ${trained}。在 Ollama 服务器上设置 OLLAMA_KV_CACHE_TYPE=q8_0 可在相同内存下大约翻倍可用上下文。` + }, providers: { connectAccount: '连接账号', haveApiKey: '改用 API 密钥?', @@ -1912,6 +1930,7 @@ export const zh: Translations = { stop: '停止', dismiss: '关闭', exit: code => `退出码 ${code}`, + ollamaLoading: (model: string) => `正在将 ${model} 加载到内存…`, coding: { title: '工作区', noBranch: '无分支', @@ -2068,6 +2087,12 @@ export const zh: Translations = { connected: '已连接', featuredPitch: '一个订阅,300+ 前沿模型 — 运行 Hermes 的推荐方式', openRouterPitch: '一个密钥,数百个模型 — 稳妥的默认选择', + detectedLocal: { + detected: '已检测到', + modelsInstalled: (n: number) => `已安装 ${n} 个模型 — 在本机运行`, + use: '使用', + connectFailed: '无法连接到本地服务器。' + }, apiKeyOptions: { openrouter: { short: '一个密钥,多个模型', description: '用一个密钥访问数百个模型。适合新安装的默认选择。' }, openai: { short: 'GPT 级模型', description: '直接访问 OpenAI 模型。' }, @@ -2132,6 +2157,8 @@ export const zh: Translations = { noAuthenticatedProviders: '没有已认证的提供方。', pro: 'Pro', proNeedsSubscription: 'Pro 模型需要付费 Nous 订阅。', + noTools: '不支持工具', + noToolsTitle: '该模型不支持工具调用 — 大多数智能体功能将无法使用。', free: '免费', freeTier: '免费层', priceTitle: '每百万 token 的输入/输出价格' diff --git a/apps/desktop/src/store/onboarding.ts b/apps/desktop/src/store/onboarding.ts index f257af1e5fe..ef399acc1ca 100644 --- a/apps/desktop/src/store/onboarding.ts +++ b/apps/desktop/src/store/onboarding.ts @@ -782,6 +782,42 @@ export async function saveOnboardingApiKey( } } +// Configure a DETECTED local Ollama server. Unlike the generic local-endpoint +// path above, the server identity is already fingerprinted and the user has +// picked a concrete model from its installed list, so there is no probe step — +// persist provider=ollama + base_url + model and verify the runtime. +export async function saveOnboardingDetectedOllama(baseUrl: string, model: string, ctx: OnboardingContext) { + const url = baseUrl.trim() + const chosen = model.trim() + + if (!url || !chosen) { + return { ok: false, message: 'Pick a model first.' } + } + + try { + await setModelAssignment({ scope: 'main', provider: 'ollama', model: chosen, base_url: url }) + await ctx.requestGateway('reload.env').catch(() => undefined) + + const runtime = await checkRuntime(ctx, 'ollama') + + if (!runtime.ready) { + const detail = (runtime.reason ?? '').trim() + + return { ok: false, message: detail || `Saved, but Hermes still cannot reach ${url}.` } + } + + notifyReady('Ollama') + completeDesktopOnboarding() + ctx.onCompleted?.() + + return { ok: true } + } catch (error) { + notifyError(error, 'Could not save Ollama endpoint') + + return { ok: false, message: errMessage(error) } + } +} + // Configure a local / self-hosted OpenAI-compatible endpoint (vLLM, llama.cpp, // Ollama, …). Unlike API-key providers, a local endpoint is defined by its URL // and usually needs NO key. The runtime resolver reads model.base_url from diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index 9133fb4dfbf..e28e0283587 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -278,6 +278,83 @@ export interface ModelOptionProvider { export interface ModelCapabilities { fast: boolean reasoning: boolean + /** True/false when a metadata source reported tool support; absent = unknown. */ + tools?: boolean + vision?: boolean + /** Trained/served context window in tokens, when known. */ + context_length?: number + /** Local Ollama models only, from the server's native metadata. */ + parameter_size?: string + quantization?: string +} + +/** One detected local model server from GET /api/local-servers/detect. */ +export interface LocalServerInfo { + /** Fingerprinted server type ("ollama", "lm-studio", ...) or the expected + * type for an unreachable well-known port. */ + type: string + /** OpenAI-compatible endpoint (server root + /v1). */ + base_url: string + reachable: boolean + models: string[] + /** "well-known" (default port probe) or "configured" (model.base_url). */ + source: string +} + +export interface LocalServersResponse { + servers: LocalServerInfo[] +} + +/** Installed model row from GET /api/ollama/models. */ +export interface OllamaInstalledModel { + name: string + size_bytes?: number + parameter_size?: string + quantization?: string + family?: string + modified_at?: string +} + +/** Loaded model row from GET /api/ollama/models (native /api/ps). */ +export interface OllamaRunningModel { + name: string + size_bytes?: number + size_vram_bytes?: number + /** Window the model was actually loaded with. */ + context_length?: number + expires_at?: string +} + +export interface OllamaRecommendedModel { + model: string + description: string + approx_vram_gb?: number +} + +/** Loaded-vs-trained context gap that q8_0 KV quantization would close. */ +export interface OllamaKvCacheAdvisory { + model: string + loaded_context: number + trained_context: number +} + +export interface OllamaModelsResponse { + reachable: boolean + installed: OllamaInstalledModel[] + running: OllamaRunningModel[] + recommended: OllamaRecommendedModel[] + kv_cache_advisory?: null | OllamaKvCacheAdvisory +} + +export interface OllamaPullStatus { + job_id: string + model: string + /** pulling | done | error */ + status: string + detail?: string + error_message?: string + total_bytes?: number + completed_bytes?: number } export interface ModelOptionsResponse { diff --git a/hermes_cli/auth.py b/hermes_cli/auth.py index 5cf6d50186a..b203cddcc3a 100644 --- a/hermes_cli/auth.py +++ b/hermes_cli/auth.py @@ -151,6 +151,10 @@ SERVICE_PROVIDER_NAMES: Dict[str, str] = { # any remote service. LMSTUDIO_NOAUTH_PLACEHOLDER = "dummy-lm-api-key" +# Same role for local Ollama servers, which are unauthenticated by default. +# Sent only to the local server, never to any remote service. +OLLAMA_NOAUTH_PLACEHOLDER = "dummy-ollama-api-key" + # ============================================================================= # Provider Registry @@ -217,6 +221,16 @@ PROVIDER_REGISTRY: Dict[str, ProviderConfig] = { api_key_env_vars=("LM_API_KEY",), base_url_env_var="LM_BASE_URL", ), + "ollama": ProviderConfig( + id="ollama", + name="Ollama (local)", + auth_type="api_key", + # 127.0.0.1, not localhost — see _normalize_ollama_runtime_base_url. + inference_base_url="http://127.0.0.1:11434/v1", + # No key env vars: local Ollama servers are unauthenticated by default. + # A key for a reverse-proxied server can be set via model.api_key. + api_key_env_vars=(), + ), "copilot": ProviderConfig( id="copilot", name="GitHub Copilot", @@ -747,6 +761,26 @@ def _normalize_lmstudio_runtime_base_url(base_url: str) -> str: return (root or "http://127.0.0.1:1234") + "/v1" +def _normalize_ollama_runtime_base_url(base_url: str) -> str: + """Return the OpenAI-compatible local Ollama runtime base URL. + + Ollama's native management API lives under ``/api`` while its + OpenAI-compatible chat endpoint lives under ``/v1``. Users paste either + form (or the bare server root) into ``model.base_url``; normalize before + the OpenAI SDK appends ``/chat/completions``. + """ + root = str(base_url or "").strip().rstrip("/") + for suffix in ("/api", "/v1"): + if root.endswith(suffix): + root = root[: -len(suffix)].rstrip("/") + break + # IPv4 loopback, not localhost: Ollama binds 127.0.0.1 by default and + # Windows resolves localhost to ::1 first — a ~2s failed IPv6 connect on + # every chat request otherwise. + root = root.replace("://localhost", "://127.0.0.1") + return (root or "http://127.0.0.1:11434") + "/v1" + + # ============================================================================= # Error Types # ============================================================================= @@ -1675,7 +1709,7 @@ def resolve_provider( "kilo": "kilocode", "kilo-code": "kilocode", "kilo-gateway": "kilocode", "lmstudio": "lmstudio", "lm-studio": "lmstudio", "lm_studio": "lmstudio", # Local server aliases — route through the generic custom provider - "ollama": "custom", "ollama_cloud": "ollama-cloud", + "ollama": "ollama", "ollama_cloud": "ollama-cloud", "vllm": "custom", "llamacpp": "custom", "llama.cpp": "custom", "llama-cpp": "custom", } @@ -6386,6 +6420,11 @@ def resolve_api_key_provider_credentials(provider_id: str) -> Dict[str, Any]: api_key = LMSTUDIO_NOAUTH_PLACEHOLDER key_source = key_source or "default" + # Same for local Ollama servers (unauthenticated by default). + if not api_key and provider_id == "ollama": + api_key = OLLAMA_NOAUTH_PLACEHOLDER + key_source = key_source or "default" + env_url = "" if pconfig.base_url_env_var: env_url = os.getenv(pconfig.base_url_env_var, "").strip() @@ -6422,6 +6461,9 @@ def resolve_api_key_provider_credentials(provider_id: str) -> Dict[str, Any]: if provider_id == "lmstudio": base_url = _normalize_lmstudio_runtime_base_url(base_url) + if provider_id == "ollama": + base_url = _normalize_ollama_runtime_base_url(base_url) + # Last-resort guard: an API-key provider must never hand back an empty # base URL (a set-but-empty COPILOT_API_BASE_URL or similar env override # otherwise wedges chat inference — #50252). diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py index 08cfe322b87..8741a6be04f 100644 --- a/hermes_cli/inventory.py +++ b/hermes_cli/inventory.py @@ -33,6 +33,8 @@ Substrate facts (verified May 2026): from __future__ import annotations +import threading + from dataclasses import dataclass, replace from typing import Optional @@ -252,13 +254,23 @@ def build_models_payload( def _apply_capabilities(rows: list[dict]) -> None: - """Attach a ``{model: {fast, reasoning}}`` map to each provider row. + """Attach a per-model capabilities map to each provider row. + + Each entry: ``{fast, reasoning, tools?, vision?, context_length?, + parameter_size?, quantization?}``. Only ``fast`` and ``reasoning`` are + always present; the rest are included when a metadata source actually + reported them, so the UI can distinguish "no tool support" from "unknown". `fast` mirrors ``model_supports_fast_mode`` (the same gate the runtime enforces). `reasoning` comes from the models.dev catalog when known and defaults to True otherwise — the effort dial is broadly accepted and a no-op on models that ignore it, whereas hiding it from a capable-but- uncatalogued model is the worse failure. + + Local Ollama rows are enriched from the server's native ``/api/show`` + instead of models.dev: local tags are arbitrary and quant-specific + (``qwen3:27b-q4_K_M``, user Modelfile creations), so the catalog rarely + matches, while ``/api/show`` is authoritative for what is on disk. """ from hermes_cli.models import model_supports_fast_mode @@ -269,25 +281,118 @@ def _apply_capabilities(rows: list[dict]) -> None: for row in rows: slug = row.get("slug") or "" - caps: dict[str, dict[str, bool]] = {} + caps: dict[str, dict] = {} for model in row.get("models") or []: - reasoning = True + entry: dict = { + "fast": bool(model_supports_fast_mode(model)), + "reasoning": True, + } if get_model_capabilities is not None and slug: try: meta = get_model_capabilities(slug, model) if meta is not None: - reasoning = bool(meta.supports_reasoning) + entry["reasoning"] = bool(meta.supports_reasoning) + entry["tools"] = bool(meta.supports_tools) + entry["vision"] = bool(meta.supports_vision) + if meta.context_window: + entry["context_length"] = int(meta.context_window) except Exception: - reasoning = True + pass - caps[model] = { - "fast": bool(model_supports_fast_mode(model)), - "reasoning": reasoning, - } + caps[model] = entry row["capabilities"] = caps + _apply_ollama_native_capabilities(rows) + + +def _apply_ollama_native_capabilities(rows: list[dict]) -> None: + """Override local Ollama rows' capabilities with native ``/api/show`` data. + + Cache-only on the request path: the picker/settings payload must never + wait on per-model ``/api/show`` round-trips (they serialize and stall the + settings skeleton). Missing metadata is backfilled by a background thread + and appears on the next payload build. + """ + ollama_rows = [r for r in rows if str(r.get("slug", "")).lower() == "ollama"] + if not ollama_rows: + return + + import time as _time + + now = _time.time() + missing: list[tuple[str, str]] = [] + for row in ollama_rows: + base_url = str(row.get("api_url", "") or "").strip() or None + caps = row.get("capabilities") or {} + for model in row.get("models") or []: + cache_key = (base_url or "", model) + cached = _OLLAMA_META_CACHE.get(cache_key) + if cached is None or (now - cached[0]) >= _OLLAMA_META_CACHE_TTL: + missing.append(cache_key) + continue + meta = cached[1] + if not meta: + continue + entry = caps.setdefault(model, {"fast": False, "reasoning": True}) + native = meta.get("capabilities") + if isinstance(native, list): + entry["tools"] = "tools" in native + entry["vision"] = "vision" in native + entry["reasoning"] = "thinking" in native + if meta.get("context_length"): + entry["context_length"] = int(meta["context_length"]) + if meta.get("parameter_size"): + entry["parameter_size"] = meta["parameter_size"] + if meta.get("quantization"): + entry["quantization"] = meta["quantization"] + row["capabilities"] = caps + + if missing: + _start_ollama_meta_backfill(missing) + + +def _start_ollama_meta_backfill(keys: list[tuple[str, str]]) -> None: + """Fetch missing model metadata off the request path.""" + import time as _time + + with _OLLAMA_META_BACKFILL_LOCK: + todo = [k for k in keys if k not in _OLLAMA_META_IN_FLIGHT] + if not todo: + return + _OLLAMA_META_IN_FLIGHT.update(todo) + + def _backfill() -> None: + try: + from hermes_cli.models import fetch_ollama_model_metadata + + for base_url, model in todo: + try: + meta = fetch_ollama_model_metadata( + model, base_url=base_url or None, timeout=5.0 + ) + if meta: + _OLLAMA_META_CACHE[(base_url, model)] = (_time.time(), meta) + except Exception: + pass + finally: + with _OLLAMA_META_BACKFILL_LOCK: + _OLLAMA_META_IN_FLIGHT.difference_update(todo) + + threading.Thread( + target=_backfill, daemon=True, name="ollama-meta-backfill" + ).start() + + +# Metadata for an installed model changes only when the user re-pulls or +# edits a Modelfile — an hour of staleness is fine and keeps repeated picker +# opens off the per-model /api/show round-trips. +_OLLAMA_META_CACHE: dict = {} +_OLLAMA_META_CACHE_TTL = 3600.0 +_OLLAMA_META_BACKFILL_LOCK = threading.Lock() +_OLLAMA_META_IN_FLIGHT: set = set() + # ─── Internal: row post-processing ────────────────────────────────────── @@ -342,6 +447,14 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list # provider. Hide it from explicit-only pickers unless it is the # current provider (handled above). continue + if slug == "ollama": + # Local Ollama servers are keyless by design — reachability is + # the credential, and the row only exists when the server + # answered the probe. A running local server is as explicit as a + # pasted API key; is_provider_explicitly_configured() can't see + # it because there is no env var or auth-store entry to find. + kept.append(row) + continue if is_provider_explicitly_configured(slug): kept.append(row) return kept @@ -378,11 +491,16 @@ def _apply_picker_hints(rows: list[dict]) -> None: ) row["auth_type"] = auth_type row["key_env"] = key_env - row["warning"] = ( - f"paste {key_env} to activate" - if auth_type == "api_key" and key_env - else f"run `hermes model` to configure ({auth_type})" - ) + if row["slug"] == "ollama": + # Keyless local server: the fix for an unconfigured row is + # starting the server, not pasting a key. + row["warning"] = "start Ollama (localhost:11434) to activate" + else: + row["warning"] = ( + f"paste {key_env} to activate" + if auth_type == "api_key" and key_env + else f"run `hermes model` to configure ({auth_type})" + ) def _reorder_canonical(rows: list[dict]) -> list[dict]: diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 5d9ecfd2050..61522d6c400 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -1642,6 +1642,21 @@ def list_authenticated_providers( live = [current_model] curated["lmstudio"] = live + # Local Ollama server: no static catalog and no key requirement — the row + # exists when the server is reachable (or is the configured provider). + # Localhost connection-refused fails in milliseconds, so this stays cheap + # when no server is running. Base URL precedence mirrors LM Studio: + # active config's base_url (when current provider is ollama) > localhost. + if "ollama" not in curated: + from hermes_cli.models import fetch_ollama_local_models + is_current_ollama = current_provider.strip().lower() == "ollama" + ollama_base = current_base_url if is_current_ollama and current_base_url else None + ollama_live = fetch_ollama_local_models(base_url=ollama_base, timeout=1.5) + if not ollama_live and is_current_ollama and current_model: + ollama_live = [current_model] + if ollama_live or is_current_ollama: + curated["ollama"] = ollama_live + # --- 1. Check Hermes-mapped providers --- from hermes_cli.models import _AGGREGATOR_PROVIDERS as _AGG_PROVIDERS from hermes_cli.providers import ALIASES as _PROVIDER_ALIAS_TABLE @@ -1807,6 +1822,11 @@ def list_authenticated_providers( has_creds = True except Exception as exc: logger.debug("Anthropic external creds check failed: %s", exc) + # Local Ollama server: unauthenticated by design, so reachability is + # the credential. The curated probe above only adds the key when the + # server responded (or it is the configured provider). + if not has_creds and hermes_slug == "ollama": + has_creds = "ollama" in curated if not has_creds: continue diff --git a/hermes_cli/models.py b/hermes_cli/models.py index a103be30ac5..6090154ae9d 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -1036,6 +1036,7 @@ CANONICAL_PROVIDERS: list[ProviderEntry] = [ ProviderEntry("moa", "Mixture of Agents", "Mixture of Agents (named presets; aggregator acts after reference models)"), ProviderEntry("novita", "NovitaAI", "NovitaAI (Cloud: Model API, Agent Sandbox, GPU Cloud)"), ProviderEntry("lmstudio", "LM Studio", "LM Studio (Local desktop app with built-in model server)"), + ProviderEntry("ollama", "Ollama (local)", "Ollama (Local model server on localhost:11434)"), ProviderEntry("anthropic", "Anthropic", "Anthropic (Claude models via API key or Claude Code)"), ProviderEntry("openai-codex", "OpenAI Codex", "OpenAI Codex (Codex CLI via ChatGPT subscription or API key)"), ProviderEntry("openai-api", "OpenAI API", "OpenAI API (api.openai.com, API key)"), @@ -1273,7 +1274,7 @@ _PROVIDER_ALIASES = { "lmstudio": "lmstudio", "lm-studio": "lmstudio", "lm_studio": "lmstudio", - "ollama": "custom", # bare "ollama" = local; use "ollama-cloud" for cloud + "ollama": "ollama", # bare "ollama" = local server; use "ollama-cloud" for cloud "ollama_cloud": "ollama-cloud", } @@ -2367,6 +2368,17 @@ def provider_model_ids(provider: Optional[str], *, force_refresh: bool = False) live = fetch_ollama_cloud_models(force_refresh=force_refresh) if live: return live + if normalized == "ollama": + # Local Ollama server: list installed models from the native API. + # Base URL precedence: active config's base_url (when the current + # provider is ollama) > localhost default. + model_cfg = _get_model_config_dict() + cfg_provider = normalize_provider(str(model_cfg.get("provider", "") or "")) + cfg_base_url = str(model_cfg.get("base_url", "") or "").strip() if cfg_provider == "ollama" else "" + cfg_api_key = str(model_cfg.get("api_key", "") or "").strip() if cfg_provider == "ollama" else "" + return fetch_ollama_local_models( + api_key=cfg_api_key, base_url=cfg_base_url or None, timeout=3.0 + ) if normalized in ("openai", "openai-api"): api_key = os.getenv("OPENAI_API_KEY", "").strip() if api_key: @@ -2637,6 +2649,13 @@ def cached_provider_model_ids( if not normalized: return [] + # Local Ollama: never disk-cache. The installed-model list changes with + # every pull/delete/server restart, the live /api/tags fetch costs + # milliseconds (fast TCP pre-check when down), and a stale cached list + # surfaces models that no longer exist in the picker. + if normalized == "ollama": + return provider_model_ids(normalized, force_refresh=force_refresh) + cache = _load_provider_models_cache() fp = _credential_fingerprint(normalized) entry = cache.get(normalized) @@ -3157,6 +3176,388 @@ def _fetch_github_models(api_key: Optional[str] = None, timeout: float = 5.0) -> return [item.get("id", "") for item in catalog if item.get("id")] +# ── Local Ollama server (native /api metadata) ────────────────────────────── +# +# The OpenAI-compatible /v1/models endpoint returns bare model ids. Ollama's +# native API is much richer: /api/tags lists installed models with size and +# family, /api/show returns capabilities (tools/vision/thinking), parameter +# size, quantization, and context length. The picker uses this to badge +# models honestly (an agent needs tool support) — inference stays on /v1. + +def _ollama_server_root(base_url: Optional[str]) -> Optional[str]: + """Server root for Ollama's native API (strip a trailing /v1 or /api). + + ``localhost`` is rewritten to ``127.0.0.1``: Ollama binds IPv4 loopback + by default, and on Windows ``localhost`` resolves to ``::1`` first, so + every request pays a ~2s failed IPv6 connect before falling back. + """ + root = str(base_url or "").strip().rstrip("/") + if not root: + return None + for suffix in ("/api", "/v1"): + if root.endswith(suffix): + root = root[: -len(suffix)].rstrip("/") + break + if "://" not in root: + root = "http://" + root + return root.replace("://localhost", "://127.0.0.1") + + +# One failed reachability probe covers every fetcher call for a while, so an +# offline server costs the model picker a single sub-second check instead of +# an HTTP timeout per call (multiplied per model by capability enrichment). +_OLLAMA_UNREACHABLE_CACHE: dict = {} +_OLLAMA_UNREACHABLE_TTL = 30.0 + + +def _ollama_server_reachable( + server_root: str, timeout: float = 0.3, use_cache: bool = True +) -> bool: + """Fast TCP pre-check before any native-API read. + + The picker and settings pages call the fetchers below on every open, so + an unreachable server must fail in milliseconds — a raw connect either + succeeds or is refused near-instantly on localhost, and the short timeout + caps the silently-dropped case. Failures are cached briefly so bulk + payload builds don't repeat the probe per model. + + ``use_cache=False`` forces a fresh probe — for status endpoints that must + notice a server coming up immediately (a cached failure would report a + just-started server as down for the rest of the TTL). A successful fresh + probe clears the negative entry for every cached caller too. + """ + import socket + import time as _time + from urllib.parse import urlparse + + now = _time.monotonic() + if use_cache: + failed_at = _OLLAMA_UNREACHABLE_CACHE.get(server_root) + if failed_at is not None and (now - failed_at) < _OLLAMA_UNREACHABLE_TTL: + return False + + try: + parsed = urlparse(server_root) + host = parsed.hostname or "localhost" + port = parsed.port or (443 if parsed.scheme == "https" else 80) + with socket.create_connection((host, port), timeout=timeout): + pass + _OLLAMA_UNREACHABLE_CACHE.pop(server_root, None) + return True + except Exception: + _OLLAMA_UNREACHABLE_CACHE[server_root] = now + return False + + +def _ollama_request_headers(api_key: Optional[str] = None) -> dict: + headers = {"Content-Type": "application/json"} + key = str(api_key or "").strip() + if key: + headers["Authorization"] = f"Bearer {key}" + return headers + + +def fetch_ollama_local_models( + api_key: Optional[str] = None, + base_url: Optional[str] = None, + timeout: float = 5.0, +) -> list[str]: + """List installed models from a local Ollama server's native ``/api/tags``. + + Returns model names (``model:tag``) or an empty list when the server is + unreachable or the response is malformed. + """ + server_root = _ollama_server_root(base_url or "http://localhost:11434") + if not server_root: + return [] + if not _ollama_server_reachable(server_root): + return [] + + request = urllib.request.Request( + server_root + "/api/tags", headers=_ollama_request_headers(api_key) + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + payload = json.loads(resp.read().decode()) + except Exception: + return [] + + models = payload.get("models") + if not isinstance(models, list): + return [] + + names: list[str] = [] + for raw in models: + if not isinstance(raw, dict): + continue + name = str(raw.get("name") or raw.get("model") or "").strip() + if name and name not in names: + names.append(name) + return names + + +def fetch_ollama_model_metadata( + model: str, + base_url: Optional[str] = None, + api_key: Optional[str] = None, + timeout: float = 5.0, +) -> Optional[dict]: + """Fetch one model's metadata from a local Ollama server's ``/api/show``. + + Returns a dict with (all fields best-effort, absent keys omitted): + - ``capabilities``: list[str] — e.g. ["completion", "tools", "vision", "thinking"] + - ``parameter_size``: str — e.g. "27B" + - ``quantization``: str — e.g. "Q4_K_M" + - ``context_length``: int — trained context from GGUF metadata + - ``family``: str — model family, e.g. "qwen3" + + Returns None when the server is unreachable or the model is unknown. + """ + server_root = _ollama_server_root(base_url or "http://localhost:11434") + if not server_root or not str(model or "").strip(): + return None + if not _ollama_server_reachable(server_root): + return None + + body = json.dumps({"name": str(model).strip()}).encode() + request = urllib.request.Request( + server_root + "/api/show", + data=body, + headers=_ollama_request_headers(api_key), + method="POST", + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + payload = json.loads(resp.read().decode()) + except Exception: + return None + if not isinstance(payload, dict): + return None + + meta: dict = {} + + caps = payload.get("capabilities") + if isinstance(caps, list): + meta["capabilities"] = [str(c).strip().lower() for c in caps if isinstance(c, str)] + + details = payload.get("details") + if isinstance(details, dict): + if details.get("parameter_size"): + meta["parameter_size"] = str(details["parameter_size"]).strip() + if details.get("quantization_level"): + meta["quantization"] = str(details["quantization_level"]).strip() + if details.get("family"): + meta["family"] = str(details["family"]).strip() + + # Context length lives in model_info under ".context_length". + model_info = payload.get("model_info") + if isinstance(model_info, dict): + for key, value in model_info.items(): + if str(key).endswith(".context_length") and isinstance(value, (int, float)): + meta["context_length"] = int(value) + break + + return meta or None + + +# Curated models that work well as Hermes agent backends on a local Ollama +# server. Every entry supports tool calling. ``approx_vram_gb`` is the rough +# working-set need at the default quantization — used to annotate the +# recommendation (never to gate it: Ollama falls back to partial offload). +OLLAMA_RECOMMENDED_MODELS: list[dict] = [ + {"model": "qwen3:8b", "description": "Qwen3 8B — tools + thinking, good starter", "approx_vram_gb": 8}, + {"model": "gpt-oss:20b", "description": "OpenAI open-weight 20B — tools + thinking", "approx_vram_gb": 16}, + {"model": "qwen3:32b", "description": "Qwen3 32B — strong agent at 24GB", "approx_vram_gb": 24}, + {"model": "mistral-small3.2:24b", "description": "Mistral Small 3.2 — tools + vision", "approx_vram_gb": 20}, + {"model": "llama3.3:70b", "description": "Llama 3.3 70B — tools, needs 48GB+", "approx_vram_gb": 48}, + {"model": "gpt-oss:120b", "description": "OpenAI open-weight 120B — tools + thinking, needs 80GB", "approx_vram_gb": 80}, +] + + +def fetch_ollama_installed_model_details( + api_key: Optional[str] = None, + base_url: Optional[str] = None, + timeout: float = 5.0, +) -> list[dict]: + """Installed models with size/details from native ``/api/tags``. + + Returns ``[{name, size_bytes, parameter_size, quantization, family, + modified_at}]`` (absent fields omitted), or ``[]`` when unreachable. + """ + server_root = _ollama_server_root(base_url or "http://localhost:11434") + if not server_root: + return [] + if not _ollama_server_reachable(server_root): + return [] + + request = urllib.request.Request( + server_root + "/api/tags", headers=_ollama_request_headers(api_key) + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + payload = json.loads(resp.read().decode()) + except Exception: + return [] + + models = payload.get("models") + if not isinstance(models, list): + return [] + + out: list[dict] = [] + for raw in models: + if not isinstance(raw, dict): + continue + name = str(raw.get("name") or raw.get("model") or "").strip() + if not name: + continue + entry: dict = {"name": name} + if isinstance(raw.get("size"), (int, float)): + entry["size_bytes"] = int(raw["size"]) + if raw.get("modified_at"): + entry["modified_at"] = str(raw["modified_at"]) + details = raw.get("details") + if isinstance(details, dict): + if details.get("parameter_size"): + entry["parameter_size"] = str(details["parameter_size"]).strip() + if details.get("quantization_level"): + entry["quantization"] = str(details["quantization_level"]).strip() + if details.get("family"): + entry["family"] = str(details["family"]).strip() + out.append(entry) + return out + + +def fetch_ollama_running_models( + api_key: Optional[str] = None, + base_url: Optional[str] = None, + timeout: float = 5.0, +) -> list[dict]: + """Loaded models from native ``/api/ps``. + + Returns ``[{name, size_bytes, size_vram_bytes, context_length, + expires_at}]`` (absent fields omitted), or ``[]`` when unreachable or + nothing is loaded. ``context_length`` is the window the model was + actually loaded with — the input for the KV-cache advisory. + """ + server_root = _ollama_server_root(base_url or "http://localhost:11434") + if not server_root: + return [] + if not _ollama_server_reachable(server_root): + return [] + + request = urllib.request.Request( + server_root + "/api/ps", headers=_ollama_request_headers(api_key) + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + payload = json.loads(resp.read().decode()) + except Exception: + return [] + + models = payload.get("models") + if not isinstance(models, list): + return [] + + out: list[dict] = [] + for raw in models: + if not isinstance(raw, dict): + continue + name = str(raw.get("name") or raw.get("model") or "").strip() + if not name: + continue + entry: dict = {"name": name} + if isinstance(raw.get("size"), (int, float)): + entry["size_bytes"] = int(raw["size"]) + if isinstance(raw.get("size_vram"), (int, float)): + entry["size_vram_bytes"] = int(raw["size_vram"]) + if isinstance(raw.get("context_length"), (int, float)): + entry["context_length"] = int(raw["context_length"]) + if raw.get("expires_at"): + entry["expires_at"] = str(raw["expires_at"]) + out.append(entry) + return out + + +def delete_ollama_model( + model: str, + api_key: Optional[str] = None, + base_url: Optional[str] = None, + timeout: float = 15.0, +) -> tuple[bool, str]: + """Delete an installed model via native ``DELETE /api/delete``. + + Returns ``(ok, message)`` — message is empty on success. + """ + server_root = _ollama_server_root(base_url or "http://localhost:11434") + name = str(model or "").strip() + if not server_root or not name: + return False, "missing model or server" + + body = json.dumps({"model": name}).encode() + request = urllib.request.Request( + server_root + "/api/delete", + data=body, + headers=_ollama_request_headers(api_key), + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + resp.read() + return True, "" + except urllib.error.HTTPError as exc: + if exc.code == 404: + return False, f"model {name!r} not found" + return False, f"delete failed with HTTP {exc.code}" + except Exception: + return False, "could not reach the Ollama server" + + +def preload_ollama_model( + model: str, + keep_alive: Optional[str] = None, + api_key: Optional[str] = None, + base_url: Optional[str] = None, + timeout: float = 300.0, +) -> tuple[bool, str]: + """Load a model into memory via a prompt-less native ``/api/generate``. + + Ollama loads the model and returns without generating. ``keep_alive`` + (e.g. ``"5m"``, ``"2h"``, ``"-1"`` for indefinite) pins how long it stays + resident; Ollama applies keep_alive per-load, so re-issue this call to + change it. The generous timeout covers cold weight loads on big models. + + Returns ``(ok, message)`` — message is empty on success. + """ + server_root = _ollama_server_root(base_url or "http://localhost:11434") + name = str(model or "").strip() + if not server_root or not name: + return False, "missing model or server" + + req_body: dict = {"model": name} + ka = str(keep_alive or "").strip() + if ka: + # Ollama accepts a Go duration string or a number of seconds; + # "-1" means keep loaded indefinitely. + req_body["keep_alive"] = int(ka) if ka.lstrip("-").isdigit() else ka + + request = urllib.request.Request( + server_root + "/api/generate", + data=json.dumps(req_body).encode(), + headers=_ollama_request_headers(api_key), + method="POST", + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + resp.read() + return True, "" + except urllib.error.HTTPError as exc: + if exc.code == 404: + return False, f"model {name!r} not found — pull it first" + return False, f"load failed with HTTP {exc.code}" + except Exception: + return False, "could not reach the Ollama server" + + _COPILOT_MODEL_ALIASES = { "openai/gpt-5": "gpt-5-mini", "openai/gpt-5-chat": "gpt-5-mini", diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 0c2a4518315..2ffe3bee6e5 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -201,6 +201,14 @@ HERMES_OVERLAYS: Dict[str, HermesOverlay] = { base_url_override="https://ollama.com/v1", base_url_env_var="OLLAMA_BASE_URL", ), + # Local Ollama server. Key optional (most local servers are unauthenticated); + # model.base_url in config overrides the default for remote boxes. + # 127.0.0.1, not localhost: Windows resolves localhost to ::1 first and + # Ollama binds IPv4 loopback — dodge the ~2s IPv6 connect penalty. + "ollama": HermesOverlay( + transport="openai_chat", + base_url_override="http://127.0.0.1:11434/v1", + ), # Azure Foundry: supports both OpenAI-style and Anthropic-style endpoints. # The transport is determined at runtime from config.yaml model.api_mode. "azure-foundry": HermesOverlay( @@ -347,7 +355,7 @@ ALIASES: Dict[str, str] = { "lmstudio": "lmstudio", "lm-studio": "lmstudio", "lm_studio": "lmstudio", - "ollama": "custom", # bare "ollama" = local; use "ollama-cloud" for cloud + "ollama": "ollama", # bare "ollama" = local server; use "ollama-cloud" for cloud "vllm": "local", "llamacpp": "local", "llama.cpp": "local", @@ -369,6 +377,7 @@ _LABEL_OVERRIDES: Dict[str, str] = { "gmi": "GMI Cloud", "tencent-tokenhub": "Tencent TokenHub", "lmstudio": "LM Studio", + "ollama": "Ollama (local)", "local": "Local endpoint", "bedrock": "AWS Bedrock", "ollama-cloud": "Ollama Cloud", diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index ddc7ccecd50..d5a69cea9f2 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -2034,6 +2034,8 @@ def resolve_runtime_provider( base_url = normalize_opencode_base_url(provider, api_mode, base_url) if provider == "lmstudio": base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url) + if provider == "ollama": + base_url = auth_mod._normalize_ollama_runtime_base_url(base_url) return { "provider": provider, "api_mode": api_mode, diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index db6b6a1ec8d..0690e068833 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -5345,6 +5345,434 @@ def get_recommended_default_model(provider: str = ""): return {"provider": slug, "model": "", "free_tier": None} +# Well-known local model-server ports probed by /api/local-servers/detect. +# Each entry: (server type reported by detect_local_server_type, default root). +_LOCAL_SERVER_PROBE_ROOTS: list[tuple[str, str]] = [ + ("ollama", "http://127.0.0.1:11434"), + ("lm-studio", "http://127.0.0.1:1234"), +] + +# Session-lived cache: detection results change rarely (a server starting or +# stopping), and onboarding + settings may both request them in quick +# succession. refresh=1 busts it, mirroring /api/model/options. +_local_servers_cache: dict = {"at": 0.0, "servers": None} +_LOCAL_SERVERS_CACHE_TTL = 60.0 + + +def _detect_local_servers_sync(profile: Optional[str]) -> list[dict]: + """Probe well-known local ports (plus the configured base_url) for + running model servers. + + Returns one entry per candidate root: + {type, base_url, reachable, models: [...], source} + ``base_url`` is the OpenAI-compatible endpoint (``/v1``). + ``type`` is ``detect_local_server_type()``'s fingerprint when reachable, + else the expected type for the well-known port. The response shape keeps + room for a future installed/running/managed lifecycle distinction. + """ + from agent.model_metadata import detect_local_server_type, is_local_endpoint + from hermes_cli.models import ( + _ollama_server_reachable, + fetch_ollama_local_models, + probe_lmstudio_models, + ) + + candidates: list[tuple[str, str, str]] = [ + (expected, root, "well-known") for expected, root in _LOCAL_SERVER_PROBE_ROOTS + ] + + # Also probe the configured base_url when it is local and isn't one of + # the defaults — the remote-Ollama-over-Tailscale case. Remote provider + # URLs (api.anthropic.com, ...) are never local servers; skip them. + try: + with _profile_scope(profile): + cfg = load_config() + model_cfg = cfg.get("model") or {} + cfg_url = str(model_cfg.get("base_url", "") or "").strip().rstrip("/") + if cfg_url and is_local_endpoint(cfg_url): + cfg_root = cfg_url[:-3].rstrip("/") if cfg_url.endswith("/v1") else cfg_url + known_roots = {root for _, root in _LOCAL_SERVER_PROBE_ROOTS} + if cfg_root and cfg_root not in known_roots: + candidates.append(("", cfg_root, "configured")) + except Exception: + pass + + servers: list[dict] = [] + for expected_type, root, source in candidates: + detected = None + # Fast TCP pre-check before the HTTP fingerprint: a down server must + # cost milliseconds, not detect_local_server_type's sequential HTTP + # timeouts (which stack across candidates and blow the client's + # request timeout). + if _ollama_server_reachable(root): + try: + detected = detect_local_server_type(root) + except Exception: + detected = None + + reachable = detected is not None + models: list[str] = [] + if reachable: + try: + if detected == "ollama": + models = fetch_ollama_local_models(base_url=root, timeout=2.0) + elif detected == "lm-studio": + models = probe_lmstudio_models(base_url=root, timeout=2.0) or [] + except Exception: + models = [] + + servers.append( + { + "type": detected or expected_type or "unknown", + "base_url": root + "/v1", + "reachable": reachable, + "models": models, + "source": source, + } + ) + return servers + + +@app.get("/api/local-servers/detect") +async def detect_local_servers(profile: Optional[str] = None, refresh: bool = False): + """Detect running local model servers (Ollama, LM Studio). + + Probes well-known localhost ports plus the configured ``model.base_url``, + fingerprinting each with ``detect_local_server_type()`` and listing the + installed models for recognized servers. Onboarding uses this to offer a + detected server as a provider choice instead of asking for a URL. + """ + import time as _time + + now = _time.time() + if ( + not refresh + and _local_servers_cache["servers"] is not None + and (now - _local_servers_cache["at"]) < _LOCAL_SERVERS_CACHE_TTL + ): + return {"servers": _local_servers_cache["servers"]} + + try: + servers = await asyncio.to_thread(_detect_local_servers_sync, profile) + except Exception: + _log.exception("GET /api/local-servers/detect failed") + raise HTTPException(status_code=500, detail="Local server detection failed") + + _local_servers_cache["servers"] = servers + _local_servers_cache["at"] = now + return {"servers": servers} + + +# --------------------------------------------------------------------------- +# Local Ollama model management +# --------------------------------------------------------------------------- +# +# Management actions (list/pull/delete/load) go through Ollama's native /api +# surface; chat inference stays on the OpenAI-compatible /v1 endpoint. Pull +# runs as a background job the UI polls — the same worker+poll shape as the +# OAuth device-code sessions above. + +_ollama_pull_jobs: dict = {} +_ollama_pull_jobs_lock = threading.Lock() +_OLLAMA_PULL_JOB_TTL = 3600.0 + + +class OllamaModelAction(BaseModel): + model: str = "" + keep_alive: Optional[str] = None + profile: Optional[str] = None + + +def _on_ollama_models_changed() -> None: + """Bust the cached ollama model-id list after a pull/delete. + + The picker payload reads ``cached_provider_model_ids("ollama")`` (1h disk + TTL); without this a just-pulled model doesn't appear in the model picker + until the cache expires. + """ + try: + from hermes_cli.models import clear_provider_models_cache + + clear_provider_models_cache("ollama") + except Exception: + pass + + +def _ollama_connection(profile: Optional[str]) -> tuple[str, str]: + """Resolve (base_url, api_key) for the local Ollama server. + + Uses model.base_url/api_key from config when the configured provider is + ollama; otherwise the localhost default. Returns the server root form + accepted by the hermes_cli.models helpers (they normalize /v1 away). + """ + base_url = "" + api_key = "" + try: + with _profile_scope(profile): + cfg = load_config() + model_cfg = cfg.get("model") or {} + provider = str(model_cfg.get("provider", "") or "").strip().lower() + if provider == "ollama": + base_url = str(model_cfg.get("base_url", "") or "").strip() + api_key = str(model_cfg.get("api_key", "") or "").strip() + except Exception: + pass + return base_url or "http://127.0.0.1:11434", api_key + + +@app.get("/api/ollama/models") +async def get_ollama_models(profile: Optional[str] = None): + """Installed + running models and the curated recommendations. + + Single round-trip payload for the desktop management panel: + ``{reachable, installed: [...], running: [...], recommended: [...]}``. + ``recommended`` excludes already-installed models. + """ + from hermes_cli.models import ( + OLLAMA_RECOMMENDED_MODELS, + fetch_ollama_installed_model_details, + fetch_ollama_running_models, + ) + + base_url, api_key = _ollama_connection(profile) + + # Fresh reachability probe first (bypasses the negative cache): this is + # the status endpoint the desktop polls, so it must notice a just-started + # server immediately. Success clears the cached failure, which un-blinds + # the fetchers below in the same request. + from hermes_cli.models import _ollama_server_reachable, _ollama_server_root + + root = _ollama_server_root(base_url) + reachable = bool(root) and await asyncio.to_thread( + _ollama_server_reachable, root, 0.3, False + ) + + installed: list = [] + running: list = [] + if reachable: + def _collect(): + return ( + fetch_ollama_installed_model_details( + api_key=api_key, base_url=base_url, timeout=3.0 + ), + fetch_ollama_running_models( + api_key=api_key, base_url=base_url, timeout=3.0 + ), + ) + + try: + installed, running = await asyncio.to_thread(_collect) + except Exception: + _log.exception("GET /api/ollama/models failed") + raise HTTPException( + status_code=500, detail="Failed to query the Ollama server" + ) + + installed_names = {m["name"] for m in installed} + # A recommendation is "installed" if any tag of the same base model is — + # qwen3:8b covers qwen3:8b-q4_K_M etc. + installed_bases = {n.split(":", 1)[0] for n in installed_names} + recommended = [ + r + for r in OLLAMA_RECOMMENDED_MODELS + if r["model"] not in installed_names + and r["model"].split(":", 1)[0] not in installed_bases + ] + + # KV-cache advisory (see docs/plans/2026-07-08-001): Ollama runs the KV + # cache at f16 unless the operator sets OLLAMA_KV_CACHE_TYPE, and never + # quantizes it to fit a larger window. When a loaded model runs at well + # under half its trained context, q8_0 KV would roughly double the usable + # window in the same memory — surface that once per panel load. Inference: + # we cannot read a remote server's env, only compare loaded vs trained. + kv_cache_advisory = None + try: + from hermes_cli.models import fetch_ollama_model_metadata + + for run in running: + loaded_ctx = int(run.get("context_length") or 0) + if loaded_ctx <= 0: + continue + meta = await asyncio.to_thread( + fetch_ollama_model_metadata, run["name"], base_url, api_key, 2.0 + ) + trained_ctx = int((meta or {}).get("context_length") or 0) + if trained_ctx > 0 and loaded_ctx * 2 <= trained_ctx: + kv_cache_advisory = { + "model": run["name"], + "loaded_context": loaded_ctx, + "trained_context": trained_ctx, + } + break + except Exception: + kv_cache_advisory = None + + return { + "reachable": reachable, + "installed": installed, + "running": running, + "recommended": recommended, + "kv_cache_advisory": kv_cache_advisory, + } + + +def _run_ollama_pull_job(job_id: str, model: str, base_url: str, api_key: str) -> None: + """Worker: stream native /api/pull NDJSON progress into the job record.""" + import urllib.request as _urlreq + + root = base_url.rstrip("/") + for suffix in ("/api", "/v1"): + if root.endswith(suffix): + root = root[: -len(suffix)].rstrip("/") + break + headers = {"Content-Type": "application/json"} + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + + def _update(**fields): + with _ollama_pull_jobs_lock: + job = _ollama_pull_jobs.get(job_id) + if job is not None: + job.update(fields) + + request = _urlreq.Request( + root + "/api/pull", + data=json.dumps({"model": model, "stream": True}).encode(), + headers=headers, + method="POST", + ) + try: + # No read timeout beyond connect: a pull legitimately runs for many + # minutes; progress lines keep the connection demonstrably alive. + with _urlreq.urlopen(request, timeout=30.0) as resp: + for raw_line in resp: + line = raw_line.decode("utf-8", errors="replace").strip() + if not line: + continue + try: + event = json.loads(line) + except Exception: + continue + if event.get("error"): + _update(status="error", error_message=str(event["error"])) + return + status = str(event.get("status", "") or "") + fields: dict = {"detail": status} + if isinstance(event.get("total"), (int, float)) and event["total"]: + fields["total_bytes"] = int(event["total"]) + fields["completed_bytes"] = int(event.get("completed", 0) or 0) + _update(**fields) + if status == "success": + _on_ollama_models_changed() + _update(status="done", detail="success") + return + # Stream ended without an explicit success line — treat as done only + # if the last event said so; otherwise report the ambiguity. + with _ollama_pull_jobs_lock: + job = _ollama_pull_jobs.get(job_id) + if job is not None and job.get("status") == "pulling": + job["status"] = "error" + job["error_message"] = "pull stream ended unexpectedly" + except Exception as exc: + _update(status="error", error_message=f"pull failed: {exc}") + + +@app.post("/api/ollama/pull") +async def start_ollama_pull(body: OllamaModelAction, request: Request): + """Start pulling a model from the Ollama registry. Token-protected. + + Returns ``{job_id}``; poll ``GET /api/ollama/pull/{job_id}`` for progress. + """ + _require_token(request) + model = (body.model or "").strip() + if not model: + raise HTTPException(status_code=400, detail="model is required") + + base_url, api_key = _ollama_connection(body.profile) + + now = time.time() + job_id = secrets.token_hex(8) + with _ollama_pull_jobs_lock: + # Reuse an active job for the same model instead of double-pulling. + for jid, job in _ollama_pull_jobs.items(): + if job.get("model") == model and job.get("status") == "pulling": + return {"job_id": jid} + # Expire finished records so the dict can't grow unboundedly. + for jid in [ + j + for j, job in _ollama_pull_jobs.items() + if (now - job.get("at", 0)) > _OLLAMA_PULL_JOB_TTL + ]: + _ollama_pull_jobs.pop(jid, None) + _ollama_pull_jobs[job_id] = { + "model": model, + "status": "pulling", + "detail": "starting", + "at": now, + } + + threading.Thread( + target=_run_ollama_pull_job, + args=(job_id, model, base_url, api_key), + daemon=True, + name=f"ollama-pull-{model}", + ).start() + return {"job_id": job_id} + + +@app.get("/api/ollama/pull/{job_id}") +async def poll_ollama_pull(job_id: str): + """Poll a pull job: ``{model, status, detail, total_bytes?, completed_bytes?}``. + + ``status``: pulling | done | error. + """ + with _ollama_pull_jobs_lock: + job = _ollama_pull_jobs.get(job_id) + if job is None: + raise HTTPException(status_code=404, detail="pull job not found or expired") + return {"job_id": job_id, **{k: v for k, v in job.items() if k != "at"}} + + +@app.post("/api/ollama/delete") +async def delete_ollama_model_endpoint(body: OllamaModelAction, request: Request): + """Delete an installed model. Token-protected.""" + _require_token(request) + model = (body.model or "").strip() + if not model: + raise HTTPException(status_code=400, detail="model is required") + + from hermes_cli.models import delete_ollama_model + + base_url, api_key = _ollama_connection(body.profile) + ok, message = await asyncio.to_thread( + delete_ollama_model, model, api_key, base_url + ) + if ok: + _on_ollama_models_changed() + return {"ok": ok, "message": message} + + +@app.post("/api/ollama/load") +async def load_ollama_model_endpoint(body: OllamaModelAction, request: Request): + """Load a model into memory (optionally pinning keep_alive). Token-protected. + + Used both for explicit warm-up from the management panel and to apply a + keep_alive change (Ollama applies keep_alive per-load). + """ + _require_token(request) + model = (body.model or "").strip() + if not model: + raise HTTPException(status_code=400, detail="model is required") + + from hermes_cli.models import preload_ollama_model + + base_url, api_key = _ollama_connection(body.profile) + ok, message = await asyncio.to_thread( + preload_ollama_model, model, body.keep_alive, api_key, base_url + ) + return {"ok": ok, "message": message} + + @app.get("/api/model/auxiliary") def get_auxiliary_models(profile: Optional[str] = None): """Return current auxiliary task assignments. diff --git a/plugins/model-providers/custom/__init__.py b/plugins/model-providers/custom/__init__.py index 2847d161adf..9712f4d8aea 100644 --- a/plugins/model-providers/custom/__init__.py +++ b/plugins/model-providers/custom/__init__.py @@ -1,9 +1,10 @@ """Custom / Ollama (local) provider profile. -Covers any endpoint registered as provider="custom", including local -Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on -Volcengine ARK, vLLM, llama.cpp). Key quirks: +Covers any endpoint registered as provider="custom", plus the first-class +"ollama" provider (routed here by alias), and OpenAI-compatible reasoning +endpoints (GLM-5.2 on Volcengine ARK, vLLM, llama.cpp). Key quirks: - ollama_num_ctx → extra_body.options.num_ctx (local context window) + - ollama_keep_alive → extra_body.keep_alive (model residence time) - reasoning_config disabled → extra_body.think = False - reasoning_config enabled + effort → top-level reasoning_effort (the native OpenAI-compatible format GLM/ARK expect; unset omits it @@ -24,6 +25,8 @@ class CustomProfile(ProviderProfile): *, reasoning_config: dict | None = None, ollama_num_ctx: int | None = None, + ollama_keep_alive: int | str | None = None, + ollama_supports_thinking: bool | None = None, **ctx: Any, ) -> tuple[dict[str, Any], dict[str, Any]]: extra_body: dict[str, Any] = {} @@ -35,6 +38,12 @@ class CustomProfile(ProviderProfile): options["num_ctx"] = ollama_num_ctx extra_body["options"] = options + # Ollama model residence time after the request (default 5m). Go + # duration string or seconds; -1 = keep loaded until server exit. + # Ignored by non-Ollama OpenAI-compatible servers. + if ollama_keep_alive is not None: + extra_body["keep_alive"] = ollama_keep_alive + # Reasoning / thinking control for custom OpenAI-compatible endpoints # (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …). # @@ -45,6 +54,10 @@ class CustomProfile(ProviderProfile): # - enabled + no effort → omit both, so the endpoint applies its own # server-side default (do NOT force a level the user didn't pick). # + # Effort levels that only exist on the OpenAI/Codex scale are mapped + # to the nearest level these endpoints accept — Ollama validates + # against high/medium/low/max/none and 400s on "xhigh"/"minimal". + # # We deliberately do NOT emit ``think=True`` on enable: it is an # Ollama-only flag and thinking is already server-default-on for these # backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't @@ -52,10 +65,19 @@ class CustomProfile(ProviderProfile): if reasoning_config and isinstance(reasoning_config, dict): _effort = (reasoning_config.get("effort") or "").strip().lower() _enabled = reasoning_config.get("enabled", True) - if _effort == "none" or _enabled is False: + if ollama_supports_thinking is False: + # The Ollama server reports this model cannot think: any + # reasoning_effort 400s ('"hermes3:8b" does not support + # thinking') and a think flag is at best a no-op. Emit + # nothing — a session's effort dial carried over from a + # thinking model must not brick the chat. None means + # unknown/non-Ollama and changes nothing. + pass + elif _effort == "none" or _enabled is False: extra_body["think"] = False elif _effort: - top_level["reasoning_effort"] = _effort + _aliases = {"xhigh": "max", "minimal": "low"} + top_level["reasoning_effort"] = _aliases.get(_effort, _effort) return extra_body, top_level diff --git a/run_agent.py b/run_agent.py index 6fb7e0903e5..f5a9b93e51f 100644 --- a/run_agent.py +++ b/run_agent.py @@ -5326,6 +5326,15 @@ class AIAgent: opts = self._lmstudio_reasoning_options_cached() # "off-only" (or absent) means no real reasoning capability. return any(opt and opt != "off" for opt in opts) + if (self.provider or "").strip().lower() == "ollama": + # Ollama rejects reasoning_effort outright on models without the + # thinking capability ('"hermes3:8b" does not support thinking'), + # so gate on the server's own /api/show capabilities. Unknown + # (old server, probe failure) errs on sending — a thinking model + # silently losing its effort dial is the worse failure, and the + # capabilities field has shipped since 0.6.0. + supported = self._ollama_supports_thinking_cached() + return supported is not False if "openrouter" not in self._base_url_lower: return False if "api.mistral.ai" in self._base_url_lower: @@ -5379,6 +5388,37 @@ class AIAgent: cache[key] = (opts, _time.monotonic()) return opts + def _ollama_supports_thinking_cached(self) -> Optional[bool]: + """Probe Ollama's /api/show thinking capability once per (model, base_url). + + True/False are cached permanently (a model's capabilities don't + change); None (probe failure / old server without the capabilities + field) is cached with a 60-second TTL so a transient failure retries + soon without an HTTP round-trip on every turn. + """ + import time as _time + + cache = getattr(self, "_ollama_thinking_cache", None) + if cache is None: + cache = self._ollama_thinking_cache = {} + key = (self.model, self.base_url) + cached = cache.get(key) + if cached is not None: + supported, ts = cached + if supported is not None or (_time.monotonic() - ts) < 60: + return supported + try: + from agent.model_metadata import query_ollama_supports_thinking + + api_key = self.api_key if isinstance(self.api_key, str) else "" + supported = query_ollama_supports_thinking( + self.model, self.base_url, api_key=api_key or "" + ) + except Exception: + supported = None + cache[key] = (supported, _time.monotonic()) + return supported + def _resolve_lmstudio_summary_reasoning_effort(self) -> Optional[str]: """Resolve a safe top-level ``reasoning_effort`` for LM Studio. diff --git a/tests/hermes_cli/test_list_picker_providers.py b/tests/hermes_cli/test_list_picker_providers.py index 480d42aa160..a3a1172da1a 100644 --- a/tests/hermes_cli/test_list_picker_providers.py +++ b/tests/hermes_cli/test_list_picker_providers.py @@ -264,6 +264,12 @@ def test_current_custom_endpoint_passthrough_marks_current_row(monkeypatch): monkeypatch.setattr("hermes_cli.providers.HERMES_OVERLAYS", {}) monkeypatch.setattr("hermes_cli.models.fetch_openrouter_models", lambda *a, **kw: []) + # The endpoint is offline in this scenario: the live /v1/models probe must + # fail so the declared model list wins. Without this mock a REAL Ollama + # server on localhost:11434 (common on dev machines) answers the probe and + # its installed models replace the declared ones. + monkeypatch.setattr("hermes_cli.models.fetch_api_models", + lambda *a, **kw: []) result = model_switch.list_picker_providers( current_provider="custom:ollama", diff --git a/tests/hermes_cli/test_model_switch_custom_providers.py b/tests/hermes_cli/test_model_switch_custom_providers.py index 4d80469b975..1cb4ba6ffba 100644 --- a/tests/hermes_cli/test_model_switch_custom_providers.py +++ b/tests/hermes_cli/test_model_switch_custom_providers.py @@ -437,6 +437,10 @@ def test_list_authenticated_providers_groups_same_endpoint(monkeypatch): returned as a single picker row with all their models merged.""" monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {}) monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {}) + # Offline endpoint: the live /v1/models probe must fail so the declared + # models win. Unmocked, a real Ollama server on localhost:11434 (common + # on dev machines) answers and its installed models replace the fixtures. + monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: []) providers = list_authenticated_providers( current_provider="custom", @@ -523,6 +527,8 @@ def test_list_authenticated_providers_distinct_endpoints_stay_separate(monkeypat even if some display names happen to be similar.""" monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {}) monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {}) + # Offline endpoints — see test_list_authenticated_providers_groups_same_endpoint. + monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: []) providers = list_authenticated_providers( user_providers={}, @@ -618,6 +624,8 @@ def test_list_authenticated_providers_total_models_reflects_grouped_count(monkey the full count, and every grouped model appears in the list.""" monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {}) monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {}) + # Offline endpoint — see test_list_authenticated_providers_groups_same_endpoint. + monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: []) entries = [ {"name": f"Ollama \u2014 Model {i}", "base_url": "http://localhost:11434/v1", diff --git a/tests/hermes_cli/test_ollama_cloud_provider.py b/tests/hermes_cli/test_ollama_cloud_provider.py index ad7e3a0b9d9..077e7a0f03a 100644 --- a/tests/hermes_cli/test_ollama_cloud_provider.py +++ b/tests/hermes_cli/test_ollama_cloud_provider.py @@ -55,14 +55,14 @@ class TestOllamaCloudAliases: """ollama_cloud (underscore) is the unambiguous cloud alias.""" assert resolve_provider("ollama_cloud") == "ollama-cloud" - def test_bare_ollama_stays_local(self): - """Bare 'ollama' alias routes to 'custom' (local) — not cloud.""" - assert resolve_provider("ollama") == "custom" + def test_bare_ollama_is_local_provider(self): + """Bare 'ollama' resolves to the local ollama provider — not cloud.""" + assert resolve_provider("ollama") == "ollama" def test_models_py_aliases(self): assert _PROVIDER_ALIASES.get("ollama_cloud") == "ollama-cloud" - # bare "ollama" stays local - assert _PROVIDER_ALIASES.get("ollama") == "custom" + # bare "ollama" is the local provider + assert _PROVIDER_ALIASES.get("ollama") == "ollama" def test_normalize_provider(self): assert normalize_provider("ollama-cloud") == "ollama-cloud" @@ -381,7 +381,7 @@ class TestOllamaCloudProvidersNew: def test_alias_resolves(self): from hermes_cli.providers import normalize_provider as np - assert np("ollama") == "custom" # bare "ollama" = local + assert np("ollama") == "ollama" # bare "ollama" = local ollama provider assert np("ollama-cloud") == "ollama-cloud" def test_label_override(self): diff --git a/tests/plugins/model_providers/test_custom_profile.py b/tests/plugins/model_providers/test_custom_profile.py index e20ee57b8ce..ac6c8d7e5a7 100644 --- a/tests/plugins/model_providers/test_custom_profile.py +++ b/tests/plugins/model_providers/test_custom_profile.py @@ -9,8 +9,8 @@ was silently dropped for every custom endpoint. These tests pin the wire-shape contract: - disabled → extra_body.think = False - enabled + effort → top-level reasoning_effort (native OpenAI-compat - format GLM/ARK expect), passed through verbatim - including ``max``/``xhigh`` + format GLM/ARK expect); OpenAI-only levels + (xhigh/minimal) map to the nearest accepted level - enabled + no effort → nothing emitted (endpoint's server default applies) - ollama_num_ctx → extra_body.options.num_ctx, orthogonal to reasoning """ @@ -65,7 +65,7 @@ class TestCustomReasoningWireShape: assert tl == {} @pytest.mark.parametrize( - "effort", ["minimal", "low", "medium", "high", "xhigh", "max"] + "effort", ["low", "medium", "high", "max"] ) def test_enabled_effort_goes_top_level(self, custom_profile, effort): """enabled + effort → TOP-LEVEL reasoning_effort, passed through verbatim. @@ -81,6 +81,26 @@ class TestCustomReasoningWireShape: assert "reasoning_effort" not in eb assert "think" not in eb + @pytest.mark.parametrize( + ("effort", "expected"), [("xhigh", "max"), ("minimal", "low")] + ) + def test_openai_only_efforts_map_to_endpoint_levels( + self, custom_profile, effort, expected + ): + """Efforts that only exist on the OpenAI/Codex scale map to the + nearest level these endpoints accept. + + Ollama validates reasoning_effort against high/medium/low/max/none + and rejects "xhigh"/"minimal" with HTTP 400; GLM documents "high" and + "max". Carrying a session's effort dial across a provider switch to a + local model must not brick the chat. + """ + eb, tl = custom_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": effort}, model="glm-5.2" + ) + assert tl == {"reasoning_effort": expected} + assert "think" not in eb + def test_enabled_without_effort_emits_nothing(self, custom_profile): """enabled but no effort → omit; do NOT force a level the user didn't pick.""" eb, tl = custom_profile.build_api_kwargs_extras(